From df20863879fe2d83077534f4726a985ffc239def Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 22 Sep 2026 10:30:02 -0300 Subject: Rename 9harness to 9agents; README, file-backed qids, worded errors - 9harness/ becomes 9agents/ (build option -D9agents, package paths). - 9agents serves /README, reports qid paths from the file's (dev, ino) and qid versions that move with the file, and names its refusals. --- 9agents/build.zig | 54 ++ 9agents/docs/DESIGN.md | 515 ++++++++++ 9agents/src/active.zig | 1006 ++++++++++++++++++++ 9agents/src/main.zig | 382 ++++++++ 9agents/src/tree.zig | 2413 +++++++++++++++++++++++++++++++++++++++++++++++ 9agents/test/e2e.sh | 358 +++++++ 9harness/build.zig | 54 -- 9harness/docs/DESIGN.md | 441 --------- 9harness/src/active.zig | 1006 -------------------- 9harness/src/main.zig | 379 -------- 9harness/src/tree.zig | 2197 ------------------------------------------ 9harness/test/e2e.sh | 347 ------- build.zig | 16 +- build.zig.zon | 4 +- 14 files changed, 4738 insertions(+), 4434 deletions(-) create mode 100644 9agents/build.zig create mode 100644 9agents/docs/DESIGN.md create mode 100644 9agents/src/active.zig create mode 100644 9agents/src/main.zig create mode 100644 9agents/src/tree.zig create mode 100755 9agents/test/e2e.sh delete mode 100644 9harness/build.zig delete mode 100644 9harness/docs/DESIGN.md delete mode 100644 9harness/src/active.zig delete mode 100644 9harness/src/main.zig delete mode 100644 9harness/src/tree.zig delete mode 100755 9harness/test/e2e.sh diff --git a/9agents/build.zig b/9agents/build.zig new file mode 100644 index 0000000..0a63d9b --- /dev/null +++ b/9agents/build.zig @@ -0,0 +1,54 @@ +//! Build fragment for 9agents: the coding-agents fs daemon (binary `9agents`), +//! a read-only fresh-from-disk 9P2000 view of every harness's state, +//! posted by default under the name `agents`. It is `@import`ed by the +//! root build.zig and called with the root builder, so every `b.path(...)` +//! here is relative to the cloud9 root (hence the `9agents/` prefix), +//! every option is defined by the root and every step it registers lands +//! in the root's step list under the `9agents` prefix. +//! +//! Steps: 9agents, 9agents-test, 9agents-itest. +const std = @import("std"); + +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + cloud9: *std.Build.Module, + /// The 9ns binary, for the --mntgen money shot in the integration + /// suite; null when 9ns is disabled, in which case the mntgen + /// section of the suite is skipped. + ns: ?*std.Build.Step.Compile, +}; + +pub const Artifacts = struct { + exe: *std.Build.Step.Compile, + /// `9agents-test`: the tree's unit tests (exclusions, path safety, + /// node ids, vanishing files — all against a fake HOME). + test_step: *std.Build.Step, + /// `9agents-itest`: test/e2e.sh (fixture roots + scratch registry, + /// plus the real-registry end-to-end when the `agents` name is free). + itest_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const mod = b.createModule(.{ + .root_source_file = b.path("9agents/src/main.zig"), + .target = ctx.target, + .optimize = ctx.optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + const exe = b.addExecutable(.{ .name = "9agents", .root_module = mod }); + const install = b.addInstallArtifact(exe, .{}); + b.getInstallStep().dependOn(&install.step); + b.step("9agents", "Build and install only the coding-agents fs daemon").dependOn(&install.step); + + const test_step = b.step("9agents-test", "Run the 9agents tree's unit tests (fake HOME; never the live roots)"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = mod })).step); + + const itest_step = b.step("9agents-itest", "Run 9agents/test/e2e.sh (posted name, reads, cmp of a transcript, mntgen mount, exclusions, live appends)"); + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("9agents/test/e2e.sh")); + run.addArtifactArg(exe); + if (ctx.ns) |ns| run.addArtifactArg(ns); + itest_step.dependOn(&run.step); + return .{ .exe = exe, .test_step = test_step, .itest_step = itest_step }; +} diff --git a/9agents/docs/DESIGN.md b/9agents/docs/DESIGN.md new file mode 100644 index 0000000..8d399d3 --- /dev/null +++ b/9agents/docs/DESIGN.md @@ -0,0 +1,515 @@ +# 9agents design + +9agents is the coding-agents fs daemon: a long-running 9P2000 server that +replaces the `zmxify` rc script's introspection half with a unified, +file-shaped, always-fresh view of every AI-agent harness's state on the +machine — Claude Code (`~/.claude`), Codex (`~/.codex`), omp (`~/.omp`), +hermes (`~/.hermes`) and dsh (`~/.dsh`), plus their skills. Everything is +read-only and read live from disk at request time; nothing is cached, so +a session transcript grows as its harness writes it and a new session +appears as soon as its file lands. + +One binary, hosted Linux only (it posts through `cloud9.post`). The +backend is `src/tree.zig`, a `cloud9.fs.Server` backend run by +`cloud9.serve.Runner`; the CLI is `src/main.zig`. Like its siblings +(9proc, 9ns) the build is a fragment in `build.zig` wired into the root +behind `-D9agents` (steps `9agents`, `9agents-test`, `9agents-itest`; +installed by the plain `zig build` next to the other programs). + +## The tree (v1 contract) + +``` +/ read-only root (dr-xr-xr-x) +/README the tree explained in place: what each entry is, + the /active fields, and the exclusions. Static text, + so it is the one fact needing no buffer. +/pid the daemon's pid, one line +/uptime seconds since start, one line +/claude/ + projects/ mirror of ~/.claude/projects//: one dir + per project, its *.jsonl session transcripts as plain + readable files, memory/ subdirs included + history ~/.claude/history.jsonl (global prompt history) + skills/ mirror of ~/.claude/skills +/codex/ + sessions/ mirror of ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl + session-index ~/.codex/session_index.jsonl + history ~/.codex/history.jsonl +/omp/ mirror of ~/.omp/agent (history.db, agent.db, models.db, + ... served as raw readable blobs; sqlite parsing is a + later step — v1 is files) +/hermes/ mirror of ~/.hermes (state.db raw, logs/) +/dsh/ mirror of ~/.dsh (profiles/, storages/, ... whatever it holds) +/skills/ union view: plain dirs claude/, codex/, omp/ mirroring each + harness's skills dir; a harness without one simply has no entry +``` + +Mirroring is lazy: nothing is walked at startup. A lookup, getattr or +readdir stats and lists the real files at request time. Unknown files and +directories appear as themselves, readable. A file that vanishes between +lookup and read answers ENOENT cleanly; a file that appears is visible on +the next request. + +Writes, creates, removes and setattrs answer EPERM (create/remove/wstat +are not declared as backend features, so the engine refuses them itself; +the write and setattr ops answer `E.PERM`, as does an open for write). +Modes are reported read-only regardless of the real bits: directories +`0o555`, files `0o444`. Sizes and mtimes are real. + +The engine's known fidelity gap applies: directory records carry fixed +modes and length 0, so `9p ls -l` shows placeholders; `9p stat` is the +accurate one. + +## The exclusions (the security boundary) + +The daemon must never become a credential reader for anything that +mounts it. The rule is absolute — `excluded()` in tree.zig, unit-tested — +and applies to every path component of every subtree, by lookup *and* by +readdir, so an excluded name neither resolves nor lists: + +- names containing `credentials`, `token`, `auth` or `secret` + (case-insensitive substrings; "auth" also hides "author-notes" — the + price of never guessing wrong); +- names ending `.key`, `.pem` or `.env`; +- the config/settings files of the mirrored harnesses, which may embed + API keys: `settings.json`, `settings.yaml`, `settings.local.json`, + `config.yml`, `config.yaml`, `config.toml`, `models.yml`, `models.yaml` + (Claude Code's settings.json is not in the tree anyway: /claude serves + only projects/, skills/ and history.jsonl); +- `.ssh`, and `id_rsa`/`id_dsa`/`id_ecdsa`/`id_ed25519` prefixes. + +Symlinks are never served, at any depth: they are an escape hatch around +the root pinning (a symlink out of `~/.claude` would make the daemon read +outside the five roots). A symlink answers ENOENT and is not listed. + +Path resolution is the mechanism that makes that true, and it is not +string composition. `openIn` walks from the pinned base one component at +a time, each opened with `O_NOFOLLOW`: the intermediate components with +`O_PATH|O_DIRECTORY` (so a component swapped for a symlink fails with +ENOTDIR rather than redirecting the walk) and the last with the caller's +flags. Every stat, read and readdir goes through it. Composing +`/` and opening that in one call is what the first version +did, and it was wrong in a way a test could not see: only the final +component's symlink was refused, so a name replaced between the walk and +the read — the plain TOCTOU — served bytes from outside every root, and +a directory above it could redirect the whole path at any time. `O_PATH` +also means a fifo or device left in a harness root is stat'd without its +open ever blocking; a read refuses anything that is not a regular file. + +The base itself comes from $HOME or `--root NAME=PATH` at startup and is +resolved normally (it is configuration, and may legitimately be a link). +Relative paths are composed only from remembered (root, rel) pairs plus +single-segment, engine-vetted names. Client bytes never form a path: +names with `/` or NUL are refused before the filesystem is touched, `.` +and `..` are refused as components inside the walk, and `..` as a lookup +resolves through the backend's own parent map, never the kernel's. + +## Qids, and what the protocol says they mean + +Read out of the Plan 9 source (`~/05-genizah/principia-softwarica`), not +recalled — the three rules below were all being broken. + +`intro(5)`: "The qid represents the server's unique identification for the +file being accessed: two files on the same server hierarchy are the same if +and only if their qids are the same. ... The path is an integer unique among +all files in the hierarchy. If a file is deleted and recreated with the same +name in the same directory, the old and new path components of the qids +should be different. The version is a version number for a file; typically, +it is incremented every time the file is modified." + +* **The qid path is the file's identity, not the server's bookkeeping.** The + backend's node id is a slot in the path table, and a slot freed by its last + release returns with a new generation — the guard that stops a forged node + id resolving. Reporting that as the qid path meant `9p stat` of one + unchanged file answered a *different* path every time (`...50000104`, + `...70000104`, `...90000104`), which by the rule above says "a different + file" on every walk. The qid path now comes from the file itself, the way + u9fs builds one — `qid.vers = st->st_mtime ^ (st->st_size << 8)` and a path + from the inode with the device mixed in (`u9fs.c`) — while the engine goes + on resolving by node id. `fs.Attr.path` exists for exactly this split and + the engine uses it for nothing else. The device is folded in because an + inode number is only unique within its filesystem and the five roots need + not share one. + +* **The qid version has to move when the file does.** It was 0 everywhere, + which tells every client that nothing this server holds ever changes — + precisely wrong for a tree whose whole point is transcripts that grow while + you read them. It is now `mtime ^ (size << 8)` for a file on disk (u9fs's + formula: the shift is what makes an append within one second still change + it) and a hash of the text for the fields this server derives rather than + reads, where `status` flipping `busy` to `idle` keeps its size and only the + content can carry the change. + +* **A directory's length is 0 on purpose.** `stat(5)`: "Directories and most + files representing devices have a conventional length of 0." That part of + the engine's "fidelity gap" note below is the convention, not a shortfall. + +## Errors are strings + +`error(5)` is `Rerror tag[2] ename[s]`: 9P2000 has no numeric error, and acme +names its refusals in words (`Eperm[] = "permission denied"`, `Eexist[] = +"file does not exist"`, `editors/acme/fsys.c`). A server that answers only +bare errnos throws away the one channel the protocol gives it for explaining +itself, and a move has seven different reasons to refuse that all used to +arrive as one undifferentiated `EPERM`: moves not allowed, illegal name, no +session resolved, no cwd, dsh has no resume form, nothing to resume, the name +is taken. Each says so now, and so does the over-cap listing, which used to +claim "Too many open files in system" — a true sentence about the wrong +resource. + +The errno is still carried beside the text, because the reader may be a Linux +mount rather than a person: 9ns maps the *text* back to an errno by +case-insensitive substring, first match wins (`enameToErrno`, 9ns/src/nine.zig). +So each message keeps the phrase that carries its errno — "not permitted" for +EPERM, "invalid" for EINVAL, "exists" for EEXIST, "no such" for ENOENT — and +avoids the ones that would silently retarget it: "permission" and "denied" map +to EACCES, "read-only" to EROFS, and the bare substring "fid" to EBADF. That +is why the read-only refusal reads "9agents serves a view, not a store: +writing is not permitted" and not the more obvious wording. + +Opens refused by the mode bits never reach the backend at all: the engine +checks `perm` first and answers `e_perm`, which is already acme's spelling. + +## Backend shape + +`Harness` (tree.zig) is the backend of `fs.Server(Harness, opts)` on +`serve.Runner` (main.zig): allocation-free request paths, comptime caps, +`Io` passed explicitly, static memory in .bss (the path table and the +per-connection buffers; ~3 MB of tables, untouched pages cost nothing). + +- **Node ids** pack `Node{idx: u8, kind: u8, serial: u48}` in the u64, + zmx-style. Static skeleton nodes (facts, the harness dirs, skills) are + `kind = .top`; every mirrored file is `kind = .path` with + `serial = (slot, generation)` of its table entry. Two walks of the same + file find the same entry, so their node ids — and qid paths — are + equal; an entry freed by its last release takes a new generation, so a + stale forged node id never resolves. +- **The path table** remembers at most 4096 remembered (root, rel) pairs, + each with a Wyhash of the pair for dedupe and a reference count. The + backend declares `features = .{ .references = true }`, so the engine + asks lookups for `.` too and pays every reference back with exactly one + `release`; an entry's refcount reaching zero frees it. Readdir records + name entries without references (a listing does not pin), so a later + walk of the same name finds the same entry and the same node id. A full + table answers `E.NFILE` — on a lookup, and on a listing that cannot + name all its entries (see Readdir). +- **Fresh reads**: a read opens the file, `readPositionalAll`s the window + at the request's offset, and closes it — every read is against the live + file, and the engine handles offsets, so `cat` of a 5 MB transcript + through a mntgen mount works. +- **Readdir** collects a directory's surviving names, sorts them for + cross-read cursor stability, and stages the engine's records + (`node:u64le dir:u8 len:u8 name`), skipping `req.off` records. The + comptime caps (1024 names, 128 KiB of name bytes, the path table) are + **loud**: a directory that would not fit answers `E.NFILE` instead of a + short listing, because a short listing cannot be told apart from a + small directory and is therefore a wrong answer, not a limitation. The + staging buffer filling is different and normal — it ends that read at + the last record that fit, and the next read continues from there; + staging stops at the first record that does not fit rather than packing + a later, shorter one behind it, which would drop that entry from the + listing entirely. A directory that changes between the reads of one + listing may shift its cursor — the fresh-tree trade, accepted in v1. +- **Locking**: one mutex spans `handle` and the reply (the answer's bytes + point into shared staging buffers), so the backend is serialized across + connections. v1 accepts this: it is an introspection fs, not a + throughput service. + +## CLI and process model + +``` +usage: 9agents [--unix PATH | --tcp IP:PORT | --fd N] [--no-post] + [--name NAME] [--root NAME=PATH]... +``` + +- Default: post itself under the name `agents` via + `serve.Runner.listenPosted`, so it appears at + `$XDG_RUNTIME_DIR/9p/agents` and every interactive fish (self-wrapped + in `9ns --mntgen`) sees it at `/mnt/9p/agents` with zero + configuration. `stop()` unposts; the stale-socket protocol covers a + killed daemon. +- `--unix`/`--tcp` listen forms (beside the post, or with `--no-post`), + `--name` to post under another name, `--root NAME=PATH` to pin any of + the five roots elsewhere (defaults are `$HOME/.`; the tests use + fake homes and never the live roots). +- `--fd N` serves exactly one 9P session over the connected stream on + descriptor N (socket activation, `9ns --spawn` handoffs), driving the + engine directly; it posts nothing. +- SIGTERM/SIGINT stop cleanly (unpost); SIGPIPE is ignored. +- No daemonization in v1: the daemon never forks or detaches, so it is + run under something that supervises it. On this machine that is a + systemd user service (`~/.config/systemd/user/9agents.service`, + `Restart=always`, lingering on), which is what keeps `/mnt/9p/agents` + there across logouts and reboots; a restart reclaims the posted name + through the stale-socket protocol. For a throwaway instance the zmx + way still works — `zmx run agents -d 9agents`. + +## Integration test plan (`test/e2e.sh`) + +Part A runs entirely on fixtures (a fake home pinned by `--root`, a +scratch `XDG_RUNTIME_DIR` registry, plan9port's `9p` as the client): + +1. the posted name is listed in the registry; +2. `9p ls /` shows the roots; a fixture transcript reads back + byte-identical (`cmp` against the source file); the skills union + mirrors the harness tree; +3. credentials-shaped files are unreachable by both listing and reading, + at several depths, in every root; +4. a file appended after the daemon started is visible immediately, and + a new session directory appears in the next listing; +5. the pid fact answers the daemon's pid; writes answer EPERM; +6. the `--unix` and `--fd` listen forms serve; +7. SIGTERM unposts the name. + +Part B is the money shot, read-only against the real roots and the real +`$XDG_RUNTIME_DIR/9p` — only when the `agents` name is free (a live +daemon owning the name skips it, not fails): the posted name in the real +registry, `9p` walks of the live `~/.claude`, a real transcript +byte-identical through the tree, the live `~/.claude/.credentials.json` +unreachable, the `9ns --mntgen` mount of `/mnt/9p/agents` with +`head` of `/claude/history`, and the stop unposting the real name. + +Unit tests (tree.zig) cover the exclusions (including the predicate +itself, spelled out), path-composition safety (slashes, NULs, symlinks, +the `..` chain, NOTDIR), node id packing (same file equal, distinct +files differ, the union identity, id recycling through release), the +vanishing file (ENOENT on lookup, getattr and read), EPERM on +write/setattr/open-for-write, and live visibility — all against a fake +HOME under a test tmp dir, never the real harness roots. + +## Verification + +- `zig build` (installs `9agents` beside 9ns, 9proc-demo, 9web). +- `zig build test` — the library's 80 tests stay green. +- `zig build 9agents-test` — the tree's unit tests. +- `zig build 9agents-itest` — `test/e2e.sh` (fixture + real registry). +- `zig build 9proc-check-freestanding` — the daemon is hosted-only but + must not break the root module. +- `zig build programs-test` / `programs-itest` — the umbrella steps, + which include 9agents's. + +## /active: the derived view (v2) + +v1 mirrors files. `/active` is the first *semantic* layer: one directory +per live agent, normalized across harnesses, so `ls /active` answers +"what is running right now" whatever wrote it. It is the discovery half +of `~/.local/bin/zmxify`, whose resolution ladder it ports. + +``` +/active/ + claude/ + 345104/ + pid 345104 ppid 344980 + started 1790013029 cwd /home/goblin + name goblin-e5 title the session's title + session 1f9a74f7-... model claude-opus-5[1m] + via registry zmx harness (read-write, below) + status busy only where the harness publishes one + transcript the live .jsonl, growing as it writes + agents/ /{model,transcript} + omp/ + 155574/ ... +``` + +A harness directory holds one directory per live process of it, named +by pid. `/proc` spells it that way and so does this: a compound +`claude-345104` would make you parse a name to recover `harness`, which +is already the directory above it, and `/active/omp/*` would not glob. + +There is no `updated` file. The last write to a session is the mtime of +`transcript`, which `stat` already carries; serving it again as its own +file would be the same fact twice, and the copy is the one that goes +stale. + +These are zmxify's picker columns — pid, harness, dir, age, zmx, via, +session, title — as files, which is the point: the script stops +scanning `/proc`, reading fds and querying sqlite itself and just reads +the tree. + +**`status` is a promise only some harnesses make.** It is the harness's +own word for what it is doing, and only Claude Code publishes one +(`busy`, in `sessions/.json`); for every other harness the file is +simply absent, the way a `skills` mount with no target is absent. +Deriving one from the process's `/proc` state would answer a different +question (`S` means "not on a CPU this instant", not "waiting for +you"). For every other harness the honest activity signal is the mtime +of `transcript`. `ls /active/*/*/status` tells you who publishes the +stronger answer. + +An entry is named `-`: unique, stable for the process's +life, and it sorts by harness. The pid is the identity because it is +what `/proc` and every harness's own registry agree on. + +### Finding the agents + +`/proc` is scanned for a process whose `argv[0]` basename is `omp`, +`claude`, `codex`, `hermes` or `dsh`, or a python running +`hermes_cli.main`. A command line carrying one of the daemon words +(`gateway`, `dashboard`, `mcp`, `mcp-server`, `app-server`, +`exec-server`, `serve`, `daemon`, `acp`, `ps`, `render`, `export`, +`__omp_worker_daemon_broker`) is a server or a helper, never a session. +The daemon's own ancestry is excluded, so 9agents can never list or +act on the process tree it lives in. + +Liveness is `/proc/` existing **and** its `stat` field 22 +(starttime) matching the one remembered for the slot. A pid that has +been reused is a different process and does not list; a corpse never +lists. + +### Resolving the session + +Each harness is asked in its own terms — the `fd -> dir -> db` ladder +zmxify worked out, with the route named in `via` so a wrong guess is +visible rather than silent: + +| harness | route | `via` | +|---|---|---| +| claude | `~/.claude/sessions/.json`, which the harness maintains itself: sessionId, cwd, name, status, version, and `procStart` as a pid-reuse guard | `registry` | +| omp | the transcript it holds open under `~/.omp/agent/sessions/`, newest first; else the store directory named after its cwd with `/` becoming `-` | `fd`, `dir` | +| dsh | the `session-/session.jsonl.zstd` it holds open; zstd, so there is no title | `fd` | +| codex | the rollout it holds open — `sessions/YYYY/MM/DD/rollout--.jsonl`, whose *name* carries the session id, so its sqlite is never opened | `fd` | +| hermes | nothing to resolve: see below | `none` | + +A resolved session is checked against the process's cwd before it is +believed (the head of an omp transcript carries `"cwd"`). A process +whose session does not resolve still appears, with everything `/proc` +knows and an empty `session`: the view never pretends to know what it +does not, and a move on it is refused. + +codex's id is the last 36 bytes of the rollout's name, not what +splitting on `-` gives — the timestamp in front of it holds dashes too. + +**hermes is the one harness with no answer here, and it needs none.** +Its sessions live only in `state.db`, with no per-session file to find; +but the only hermes processes that run are the gateway and the +dashboard, and both are daemon-shaped, so neither is a session anything +should move. zmxify resolved a hermes id from sqlite and then declined +to touch the process holding it, for the same reason. If an interactive +hermes ever exists, this is the gap, and it is the one place sqlite +would buy something. + +### Freshness + +A slot remembers only what identifies the agent: harness, pid, +starttime, cwd, session id and transcript path. Every *field* is +re-derived from disk when it is read, so `status` is never stale and a +transcript grows under `cat`. `/proc` is rescanned on a readdir of +`/active` and on a lookup that misses, not on every read. + +### `zmx`: the write path, and why it is not a ctl + +Writing a zmx session name into `/active//zmx` moves that agent into +a zmx session of that name: the zmxify action half, as a file. + +Reading `zmx` gives the `ZMX_SESSION` of the process, empty when it runs +outside zmx. Writing sets it. The file means *which zmx session this +agent lives in*, and writing a name makes that true — state, not a verb +channel. A `ctl` taking words would be the ordinary Plan 9 spelling +(`/proc/n/ctl`), and an executable `zmxify` script served in the tree +would be the zmx `attach` spelling, but a script that shells out to a +local binary is a lie over a remote mount: it would run against a +session that is not on the client's machine. A write is served where +the authority is, so it survives being mounted from anywhere. + +What a write does, in order, refusing before it destroys anything: + +1. The name must be zmx's label charset (`[A-Za-z0-9._-]`) and unused by + a live session, else `EEXIST`. +2. The agent's session must have resolved, else `EPERM`. Nothing is + killed that has nowhere to come back to — zmxify's invariant. +3. `SIGTERM`, then `SIGKILL` after 5s, giving up at 12s with `EIO`. +4. `fork`, `chdir` to the agent's cwd, `exec zmx run -d` with a + **fixed argv per harness** (`claude --resume `, + `omp --resume `, `codex resume `, + `hermes --resume `). No client byte ever reaches `exec`: the + only thing the client supplies is the session name, and it is + validated first. +5. The write returns once `$XDG_RUNTIME_DIR/9p/zmx/` appears + (zmx self-posts), or `EIO` on timeout. + +The agent comes back under a new pid, so the old `/active/` is gone +and a new entry takes its place with `zmx` reading the new name. The +window between the kill and the exec is the same exposure zmxify has +always had, and it is why step 2 comes first. + +**This is the tree's first write path, and it is an escalation**: a +client that can write this file can kill the user's agents and cause a +process to be spawned. It is therefore off unless `--allow-move` is +given, and the read-only daemon stays the default. Everything else in +the tree still answers `EPERM` to writes. + +### Where this is not Plan 9 + +Named, because they are choices rather than oversights: + +* **`via` describes how the server found out**, not something true of + the process. That is diagnostics in the interface. It stays because a + resolution that guesses wrong silently is worse than a wart — it is + why zmxify's picker showed the route too. +* **A write to `zmx` does not change the object, it replaces it.** The + process is killed and another starts under a new pid, so the entry + written to disappears. `/proc/n/ctl` accepting `kill` is at least + honest about being a verb with a consequence. The file still beats an + executable served in the tree, which would lie outright over a remote + mount, so it stays — with this paragraph. +* **A move blocks the whole daemon** for as long as the kill and the + restart take, because one mutex spans `handle` and the reply. A Plan + 9 server keeps answering other fids meanwhile. The engine already has + parked replies and Tflush for exactly this; the move does not use + them yet, and that is the first thing to revisit. +* **`/active` lives inside the mirror** rather than being its own + posted service. "What files exist" and "what is running" are + different jobs, and a stricter reading would separate them; they + share the pinned roots and the parsing glue, so they share a daemon. + +### Replacing zmxify + +There is no replacement script, because the mount is the replacement: + +``` +ls /mnt/9p/agents/active/*/* what is running +cat /mnt/9p/agents/active/claude/345104/session what it is +echo mine > /mnt/9p/agents/active/claude/345104/zmx +zmx attach mine +``` + +`~/.local/bin/zmxify` was 307 lines of rc because there was no +filesystem to ask: it scanned `/proc`, classified command lines, read +fd tables and queried two sqlite databases to work out what the four +lines above now read. All of that moved into the daemon, where it is +tested. Writing a wrapper back on top would only re-derive what the +tree already says, and would become a second, worse interface beside +the real one. + +Two behaviours the old script had are worth knowing, because they are +now the caller's to arrange and both are one line: pick a session name +nothing holds (the registry, `$XDG_RUNTIME_DIR/9p/zmx`, says which are +taken) and `zmx attach` afterwards. + +A third is gone rather than moved. The old script refused to act on +its own ancestry, because it did the killing itself: kill your own +parent and the script dies before it can re-exec, losing the session it +was rescuing. The daemon kills, re-execs and waits for the new session +to post, and completes all of it whether or not the client is still +connected — verified by hanging up immediately after sending the write +and watching the move land anyway. So moving the terminal you are +sitting in works, which is the common case; the shell drops, and you +reattach with the name you wrote. + +### Testability + +`--proc DIR` overrides `/proc` the way `--root NAME=PATH` overrides a +harness root, so the scan, the liveness guard, the resolution ladder and +the exclusions are unit-tested against a fixture process tree and never +against the live machine. The move itself is exercised end-to-end +against a fake harness in `test/e2e.sh`. + +## Out of scope for v1 (documented, not hidden) + +- Write paths (create/write/setattr stay EPERM), resume/attach + automation (the zmxify re-exec half), sqlite parsing (omp/hermes + sessions stay raw blobs), cross-session search, any caching or + invalidation layer. +- Streaming/parked reads (a read answers at once; the transcripts are + plain files), and union-directory semantics beyond the plain + /skills/ dirs. diff --git a/9agents/src/active.zig b/9agents/src/active.zig new file mode 100644 index 0000000..daaddc9 --- /dev/null +++ b/9agents/src/active.zig @@ -0,0 +1,1006 @@ +//! The derived view behind `/active`: one record per live agent, +//! normalized across harnesses. +//! +//! v1 of the tree mirrors files. This module answers a different +//! question — *what is running right now* — by reading `/proc` and then +//! asking each harness in its own terms which stored session the +//! process is writing. That ladder (`fd` -> `dir` -> `db`, with the +//! route named so a wrong guess is visible) is `~/.local/bin/zmxify`'s, +//! ported here so the knowledge lives somewhere tested. +//! +//! Everything here composes paths from the proc root, a pinned harness +//! root and a pid. No client byte reaches the filesystem, and the proc +//! root is a parameter (`--proc`) so the scan, the liveness guard and +//! the exclusions are testable against a fixture tree. +const std = @import("std"); +const Io = std.Io; + +// ---- comptime bounds --------------------------------------------------------- + +/// Live agents held at once. A machine running more than this many +/// interactive harnesses at the same time is not the case to optimize. +pub const max_live: usize = 32; +/// Subagents listed under one agent. +pub const max_agents: usize = 32; +/// Longest directory a process may run in. +pub const cwd_capacity: usize = 512; +/// Longest absolute path this module composes (proc root, harness root). +pub const path_capacity: usize = 1024; +/// Longest session id, title, name or model answered. +pub const text_capacity: usize = 160; +/// `-`. +pub const name_capacity: usize = 32; +/// Bytes of a `/proc` file or a transcript head read at once. +pub const probe_capacity: usize = 8192; + +/// The harnesses this view knows how to recognize. +/// The order is `tree.Root`'s, and `tree.zig` asserts that at comptime: +/// the two enums index the same pinned roots, so a mismatch would hand +/// one harness another's base path. +pub const Kind = enum(u8) { + claude, + codex, + omp, + hermes, + dsh, + + pub fn text(k: Kind) []const u8 { + return @tagName(k); + } +}; + +/// How a session was resolved — reported, so a wrong guess is visible +/// rather than silent. `none` means the process is listed with what +/// `/proc` knows and nothing more, and cannot be moved: hermes is the +/// only harness that always lands here. +pub const Via = enum(u8) { + none, + registry, + fd, + dir, + + pub fn text(v: Via) []const u8 { + return @tagName(v); + } +}; + +/// A fixed-capacity byte buffer. Everything this module remembers is +/// bounded at comptime, like the rest of the daemon. +fn Buf(comptime n: usize) type { + return struct { + bytes: [n]u8 = undefined, + len: u16 = 0, + + const Self = @This(); + + pub fn set(b: *Self, v: []const u8) void { + const take = @min(v.len, n); + @memcpy(b.bytes[0..take], v[0..take]); + b.len = @intCast(take); + } + + pub fn slice(b: *const Self) []const u8 { + return b.bytes[0..b.len]; + } + + pub fn clear(b: *Self) void { + b.len = 0; + } + }; +} + +/// One live agent. +pub const Live = struct { + kind: Kind = .claude, + pid: u32 = 0, + ppid: u32 = 0, + /// `/proc//stat` field 22. Two processes can share a pid over + /// time; they cannot share a pid and a start time, so this is what + /// makes a remembered slot safe to trust later. + starttime: u64 = 0, + /// Unix seconds, from the process's start time against boot time. + started: i64 = 0, + via: Via = .none, + name: Buf(name_capacity) = .{}, + cwd: Buf(cwd_capacity) = .{}, + session: Buf(text_capacity) = .{}, + /// Absolute path of the session transcript, empty when unresolved. + transcript: Buf(path_capacity) = .{}, + /// Directory holding the subagent transcripts, empty when the + /// harness has no such layout. + agent_dir: Buf(path_capacity) = .{}, +}; + +// ---- pure helpers: the part that carries zmxify's knowledge ------------------- + +/// A command line word that marks a daemon, a server or a helper — a +/// process wearing the harness's name that is not an interactive +/// session and must never be listed or acted on. +const not_session = [_][]const u8{ + "__omp_worker_daemon_broker", + "acp", + "mcp", + "mcp-server", + "app-server", + "exec-server", + "gateway", + "dashboard", + "serve", + "daemon", + "ps", + "render", + "export", +}; + +/// The last path component of `p`. +pub fn basename(p: []const u8) []const u8 { + if (std.mem.lastIndexOfScalar(u8, p, '/')) |i| return p[i + 1 ..]; + return p; +} + +/// Iterates a `/proc//cmdline`: NUL-separated argv, usually with a +/// trailing NUL. An empty cmdline (a kernel thread) yields nothing. +pub const Argv = struct { + rest: []const u8, + + pub fn init(cmdline: []const u8) Argv { + return .{ .rest = cmdline }; + } + + pub fn next(a: *Argv) ?[]const u8 { + while (a.rest.len > 0 and a.rest[0] == 0) a.rest = a.rest[1..]; + if (a.rest.len == 0) return null; + const end = std.mem.indexOfScalar(u8, a.rest, 0) orelse a.rest.len; + const word = a.rest[0..end]; + a.rest = a.rest[@min(end + 1, a.rest.len)..]; + return word; + } +}; + +/// The harness a command line runs, if any. `argv[0]`'s basename names +/// it, except for hermes, which runs as a python module; any word from +/// `not_session` disqualifies the whole command line. +pub fn harnessOf(cmdline: []const u8) ?Kind { + var it: Argv = .init(cmdline); + const argv0 = it.next() orelse return null; + const exe_name = basename(argv0); + + var found: ?Kind = null; + if (std.mem.eql(u8, exe_name, "claude")) { + found = .claude; + } else if (std.mem.eql(u8, exe_name, "omp")) { + found = .omp; + } else if (std.mem.eql(u8, exe_name, "codex")) { + found = .codex; + } else if (std.mem.eql(u8, exe_name, "hermes")) { + found = .hermes; + } else if (std.mem.eql(u8, exe_name, "dsh")) { + found = .dsh; + } else if (std.mem.startsWith(u8, exe_name, "python") or std.mem.startsWith(u8, exe_name, "pypy")) { + var py: Argv = .init(cmdline); + while (py.next()) |w| { + if (std.mem.endsWith(u8, w, "hermes") or std.mem.endsWith(u8, w, "hermes_cli.main")) { + found = .hermes; + break; + } + } + } + if (found == null) return null; + + var words: Argv = .init(cmdline); + while (words.next()) |w| { + for (not_session) |bad| if (std.mem.eql(u8, w, bad)) return null; + } + return found; +} + +/// Field `n` (1-based) of a `/proc//stat` line. Field 2 is the +/// executable name in parentheses and may itself hold spaces and +/// parentheses, so the fields are counted from the *last* `)`, never by +/// splitting the whole line. Only fields from 3 on are reachable, which +/// is all any caller wants. +pub fn statField(stat: []const u8, n: usize) ?u64 { + if (n < 3) return null; + const close = std.mem.lastIndexOfScalar(u8, stat, ')') orelse return null; + var rest = stat[close + 1 ..]; + var field: usize = 3; // state, the first field after the name + while (true) { + while (rest.len > 0 and rest[0] == ' ') rest = rest[1..]; + if (rest.len == 0) return null; + const end = std.mem.indexOfScalar(u8, rest, ' ') orelse rest.len; + if (field == n) return std.fmt.parseInt(u64, rest[0..end], 10) catch null; + rest = rest[end..]; + field += 1; + } +} + +/// The process's start time in clock ticks since boot. Two processes can +/// share a pid over time; they cannot share a pid and a start time, so +/// this is what makes a remembered slot safe to trust later. +pub fn startTimeOf(stat: []const u8) ?u64 { + return statField(stat, 22); +} + +/// The parent pid. +pub fn parentOf(stat: []const u8) ?u32 { + const v = statField(stat, 4) orelse return null; + return std.math.cast(u32, v); +} + +/// The value of `"key"` in a small flat JSON object: a quoted string +/// without its quotes, or a bare token (a number, `true`, `null`). +/// Nested objects are not searched, which is what the callers want — +/// these are the harnesses' own one-level session records. +pub fn jsonField(text: []const u8, key: []const u8) ?[]const u8 { + var quoted_buf: [64]u8 = undefined; + if (key.len + 2 > quoted_buf.len) return null; + quoted_buf[0] = '"'; + @memcpy(quoted_buf[1 .. 1 + key.len], key); + quoted_buf[1 + key.len] = '"'; + const quoted = quoted_buf[0 .. key.len + 2]; + + var from: usize = 0; + while (std.mem.indexOfPos(u8, text, from, quoted)) |at| { + from = at + quoted.len; + var rest = text[from..]; + while (rest.len > 0 and (rest[0] == ' ' or rest[0] == '\t')) rest = rest[1..]; + if (rest.len == 0 or rest[0] != ':') continue; + rest = rest[1..]; + while (rest.len > 0 and (rest[0] == ' ' or rest[0] == '\t')) rest = rest[1..]; + if (rest.len == 0) return null; + if (rest[0] == '"') { + const end = std.mem.indexOfScalar(u8, rest[1..], '"') orelse return null; + return rest[1 .. 1 + end]; + } + if (rest[0] == '{' or rest[0] == '[') continue; // not a leaf; keep looking + const end = std.mem.indexOfAny(u8, rest, ",}\n\r \t") orelse rest.len; + if (end == 0) return null; + return rest[0..end]; + } + return null; +} + +/// The session id in a transcript's file name. omp writes +/// `_.jsonl`, so the id is what follows the last `_`. +/// codex writes `rollout--.jsonl`, where the +/// timestamp holds dashes too, so the id is the last 36 bytes rather +/// than anything found by splitting. Claude Code writes +/// `.jsonl`, all of it the id. +pub fn sessionIdOf(kind: Kind, file_name: []const u8) []const u8 { + var stem = file_name; + if (std.mem.endsWith(u8, stem, ".jsonl")) stem = stem[0 .. stem.len - ".jsonl".len]; + switch (kind) { + .omp => if (std.mem.lastIndexOfScalar(u8, stem, '_')) |i| return stem[i + 1 ..], + .codex => { + const uuid_len = "01a0bcd2-5653-7cc0-aa36-676f6467a72a".len; + if (stem.len >= uuid_len) return stem[stem.len - uuid_len ..]; + }, + else => {}, + } + return stem; +} + +/// The store directory a harness derives from a working directory. +/// omp strips `$HOME` and turns every `/` into `-`; Claude Code turns +/// every character that is not a letter or a digit into `-`, over the +/// whole path. Returns the slug written into `out`. +pub fn slugOf(kind: Kind, cwd: []const u8, home: []const u8, out: []u8) ?[]const u8 { + var path = cwd; + if (kind == .omp and home.len > 0 and std.mem.startsWith(u8, path, home)) { + path = path[home.len..]; + } + if (path.len > out.len) return null; + for (path, 0..) |c, i| { + out[i] = switch (kind) { + .omp => if (c == '/') '-' else c, + else => if (std.ascii.isAlphanumeric(c)) c else '-', + }; + } + return out[0..path.len]; +} + +/// Is `name` a zmx session name? zmx allows only these bytes in a +/// label, and the move path never composes anything else. +pub fn legalZmxName(name: []const u8) bool { + if (name.len == 0 or name.len > 64) return false; + for (name) |c| { + if (!std.ascii.isAlphanumeric(c) and c != '.' and c != '_' and c != '-') return false; + } + return true; +} + + +// ---- reading /proc and the harness stores ------------------------------------ + +/// Where the view reads from. Every path below is composed from these +/// and a pid; nothing here is ever built from a client's bytes. +pub const Sources = struct { + io: Io, + /// The proc filesystem root — `/proc`, or a fixture tree under test. + proc: []const u8, + /// $HOME, for the store-slug spellings. + home: []const u8, + /// The pinned harness roots, indexed by `@intFromEnum(Kind)`; an + /// empty base means that harness is unreachable. + roots: [5][]const u8 = @splat(""), + /// The daemon's own pid: it and its ancestors are never listed, so + /// 9agents can neither show nor act on the tree it lives in. Zero + /// disables the check (fixtures have no ancestry). + self_pid: u32 = 0, + /// Seconds between the epoch and boot, from `/proc/stat`'s `btime`. + /// Zero when it could not be read; `started` is then zero too. + btime: i64 = 0, +}; + +/// USER_HZ: the unit of `/proc//stat`'s time fields. Linux reports +/// them in hundredths of a second whatever CONFIG_HZ is. +const user_hz: u64 = 100; + +fn readSmall(io: Io, path: []const u8, buf: []u8) ?[]const u8 { + const out = Io.Dir.readFile(.cwd(), io, path, buf) catch return null; + return out; +} + +fn linkOf(io: Io, path: []const u8, buf: []u8) ?[]const u8 { + const n = Io.Dir.readLinkAbsolute(io, path, buf) catch return null; + return buf[0..n]; +} + +fn statOf(io: Io, path: []const u8) ?Io.Dir.Stat { + return Io.Dir.statFile(.cwd(), io, path, .{ .follow_symlinks = true }) catch null; +} + +fn mtimeOf(io: Io, path: []const u8) i64 { + const st = statOf(io, path) orelse return 0; + return st.mtime.toSeconds(); +} + +/// `parts` joined with `/` into `buf`, or null when they do not fit. +fn join(buf: []u8, parts: []const []const u8) ?[]const u8 { + var n: usize = 0; + for (parts, 0..) |p, i| { + if (i > 0) { + if (n + 1 > buf.len) return null; + buf[n] = '/'; + n += 1; + } + if (n + p.len > buf.len) return null; + @memcpy(buf[n..][0..p.len], p); + n += p.len; + } + return buf[0..n]; +} + +/// `` of a harness, or null when it is not pinned. +fn rootOf(s: Sources, k: Kind) ?[]const u8 { + const base = s.roots[@intFromEnum(k)]; + return if (base.len == 0) null else base; +} + +/// The directory a harness keeps its session transcripts under. +fn sessionRoot(s: Sources, k: Kind, buf: []u8) ?[]const u8 { + const base = rootOf(s, k) orelse return null; + return switch (k) { + // The omp root is ~/.omp and its mirror hangs off agent/. + .omp => join(buf, &.{ base, "agent", "sessions" }), + .claude => join(buf, &.{ base, "projects" }), + .dsh => join(buf, &.{ base, "sessions" }), + // codex names each rollout after its session, so the file it + // holds open is the whole answer — no sqlite needed. + .codex => join(buf, &.{ base, "sessions" }), + // hermes keeps its sessions only in sqlite, and the only hermes + // processes that run are the gateway and the dashboard, which + // are never sessions. Nothing to resolve. + .hermes => null, + }; +} + +// ---- the table ---------------------------------------------------------------- + +pub const Slot = struct { + used: bool = false, + /// Bumped whenever the slot comes to hold a different process, so a + /// node id minted for the old one stops resolving. + gen: u8 = 0, + seen: bool = false, + live: Live = .{}, +}; + +pub const Table = struct { + slots: [max_live]Slot = @splat(.{}), + + /// The slot holding this (pid, starttime), if any. + fn slotOfPid(t: *Table, pid: u32, starttime: u64) ?*Slot { + for (&t.slots) |*sl| { + if (sl.used and sl.live.pid == pid and sl.live.starttime == starttime) return sl; + } + return null; + } + + fn freeSlot(t: *Table) ?*Slot { + for (&t.slots) |*sl| if (!sl.used) return sl; + return null; + } + + /// The live record at `index`, when its generation still matches. + pub fn at(t: *Table, index: usize, gen: u8) ?*Live { + if (index >= t.slots.len) return null; + const sl = &t.slots[index]; + if (!sl.used or sl.gen != gen) return null; + return &sl.live; + } + + pub fn indexOf(t: *Table, l: *const Live) usize { + const base = @intFromPtr(&t.slots[0]); + return (@intFromPtr(l) - @sizeOf(Slot) + @sizeOf(Slot) - @offsetOf(Slot, "live") - base) / @sizeOf(Slot); + } + + /// Rescans `/proc`. Slots keep their index and generation while the + /// same process is still there, so a handle opened before the scan + /// keeps working; a slot that comes to hold a different process + /// bumps its generation and the old node ids stop resolving. + pub fn scan(t: *Table, s: Sources) void { + for (&t.slots) |*sl| sl.seen = false; + + var anc: [64]u32 = @splat(0); + const anc_n = ancestry(s, &anc); + + var buf: [path_capacity]u8 = undefined; + if (Io.Dir.openDirAbsolute(s.io, s.proc, .{ .iterate = true })) |dir| { + defer Io.Dir.close(dir, s.io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(dir, &rb); + while (true) { + const entry = (it.next(s.io) catch break) orelse break; + const pid = std.fmt.parseInt(u32, entry.name, 10) catch continue; + var skip = false; + for (anc[0..anc_n]) |a| if (a == pid) { + skip = true; + }; + if (skip) continue; + t.consider(s, pid, &buf); + } + } else |_| {} + + for (&t.slots) |*sl| { + if (sl.used and !sl.seen) { + sl.used = false; + sl.gen +%= 1; + } + } + } + + /// Looks at one pid and takes it if it is an agent. + fn consider(t: *Table, s: Sources, pid: u32, buf: []u8) void { + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch return; + var dir_buf: [path_capacity]u8 = undefined; + const pid_dir = join(&dir_buf, &.{ s.proc, pid_text }) orelse return; + + var stat_buf: [probe_capacity]u8 = undefined; + const stat_path = join(buf, &.{ pid_dir, "stat" }) orelse return; + const stat = readSmall(s.io, stat_path, &stat_buf) orelse return; + const starttime = startTimeOf(stat) orelse return; + + var cmd_buf: [probe_capacity]u8 = undefined; + const cmd_path = join(buf, &.{ pid_dir, "cmdline" }) orelse return; + const cmdline = readSmall(s.io, cmd_path, &cmd_buf) orelse return; + const kind = harnessOf(cmdline) orelse return; + + // Already known and still the same process: keep the slot and + // everything resolved for it. + if (t.slotOfPid(pid, starttime)) |sl| { + sl.seen = true; + return; + } + + const sl = t.freeSlot() orelse return; // full: the rest simply do not list + sl.* = .{ .used = true, .gen = sl.gen +% 1, .seen = true, .live = .{} }; + const l = &sl.live; + l.kind = kind; + l.pid = pid; + l.ppid = parentOf(stat) orelse 0; + l.starttime = starttime; + l.started = if (s.btime == 0) 0 else s.btime + @as(i64, @intCast(starttime / user_hz)); + + var name_buf: [name_capacity]u8 = undefined; + l.name.set(std.fmt.bufPrint(&name_buf, "{s}-{d}", .{ kind.text(), pid }) catch kind.text()); + + const cwd_path = join(buf, &.{ pid_dir, "cwd" }) orelse return; + var cwd_buf: [cwd_capacity]u8 = undefined; + if (linkOf(s.io, cwd_path, &cwd_buf)) |cwd| l.cwd.set(cwd); + + resolveSession(l, s, pid_dir); + } +}; + +/// The daemon's own pid and every parent of it, so the tree it lives in +/// is never listed. Returns how many were written. +fn ancestry(s: Sources, out: []u32) usize { + if (s.self_pid == 0) return 0; + var n: usize = 0; + var pid = s.self_pid; + var buf: [path_capacity]u8 = undefined; + var stat_buf: [probe_capacity]u8 = undefined; + while (pid > 1 and n < out.len) { + out[n] = pid; + n += 1; + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch break; + const path = join(&buf, &.{ s.proc, pid_text, "stat" }) orelse break; + const stat = readSmall(s.io, path, &stat_buf) orelse break; + const parent = parentOf(stat) orelse break; + if (parent == 0 or parent == pid) break; + pid = parent; + } + return n; +} + +/// Asks the harness, in its own terms, which stored session this +/// process is writing. Leaves `via = .none` when it cannot tell: the +/// view never invents a session it did not find. +fn resolveSession(l: *Live, s: Sources, pid_dir: []const u8) void { + switch (l.kind) { + .claude => resolveClaude(l, s), + .omp => resolveOmp(l, s, pid_dir), + .dsh, .codex => resolveByFd(l, s, pid_dir), + // hermes has no per-session file to find. + .hermes => {}, + } +} + +/// Claude Code maintains `sessions/.json` itself, so there is +/// nothing to guess — only to check. `procStart` must match the +/// process's start time, or the record belongs to a dead predecessor +/// that happened to share the pid. +fn resolveClaude(l: *Live, s: Sources) void { + const base = rootOf(s, .claude) orelse return; + var buf: [path_capacity]u8 = undefined; + var num: [24]u8 = undefined; + const file = std.fmt.bufPrint(&num, "{d}.json", .{l.pid}) catch return; + const path = join(&buf, &.{ base, "sessions", file }) orelse return; + var text_buf: [probe_capacity]u8 = undefined; + const text = readSmall(s.io, path, &text_buf) orelse return; + + const proc_start = jsonField(text, "procStart") orelse return; + const claimed = std.fmt.parseInt(u64, proc_start, 10) catch return; + if (claimed != l.starttime) return; // a record left by a reused pid + + const sid = jsonField(text, "sessionId") orelse return; + if (sid.len == 0 or sid.len > text_capacity) return; + l.session.set(sid); + l.via = .registry; + + var slug_buf: [cwd_capacity]u8 = undefined; + const slug = slugOf(.claude, l.cwd.slice(), s.home, &slug_buf) orelse return; + var tr_buf: [path_capacity]u8 = undefined; + var file_buf: [text_capacity + 8]u8 = undefined; + const tr_name = std.fmt.bufPrint(&file_buf, "{s}.jsonl", .{sid}) catch return; + const tr = join(&tr_buf, &.{ base, "projects", slug, tr_name }) orelse return; + if (statOf(s.io, tr) != null) l.transcript.set(tr); +} + +/// omp: the transcript it holds open, else the store directory named +/// after its working directory. +fn resolveOmp(l: *Live, s: Sources, pid_dir: []const u8) void { + resolveByFd(l, s, pid_dir); + if (l.via == .none) resolveOmpByDir(l, s); + if (l.transcript.len == 0) return; + const tr = l.transcript.slice(); + l.session.set(sessionIdOf(.omp, basename(tr))); + // The subagents are sibling files in the directory named after the + // session file, one `.jsonl` each. + if (std.mem.endsWith(u8, tr, ".jsonl")) { + l.agent_dir.set(tr[0 .. tr.len - ".jsonl".len]); + } +} + +fn resolveOmpByDir(l: *Live, s: Sources) void { + var root_buf: [path_capacity]u8 = undefined; + const root = sessionRoot(s, .omp, &root_buf) orelse return; + var slug_buf: [cwd_capacity]u8 = undefined; + const slug = slugOf(.omp, l.cwd.slice(), s.home, &slug_buf) orelse return; + var dir_buf: [path_capacity]u8 = undefined; + const dir_path = join(&dir_buf, &.{ root, slug }) orelse return; + var newest_buf: [path_capacity]u8 = undefined; + const newest = newestUnder(s.io, dir_path, ".jsonl", &newest_buf) orelse return; + l.transcript.set(newest); + l.via = .dir; +} + +/// The newest entry of `dir_path` whose name ends in `suffix`, as a +/// full path written into `out`. +fn newestUnder(io: Io, dir_path: []const u8, suffix: []const u8, out: []u8) ?[]const u8 { + const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch return null; + defer Io.Dir.close(dir, io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(dir, &rb); + var best: i64 = -1; + var best_len: usize = 0; + var probe: [path_capacity]u8 = undefined; + while (true) { + const entry = (it.next(io) catch break) orelse break; + if (!std.mem.endsWith(u8, entry.name, suffix)) continue; + const path = join(&probe, &.{ dir_path, entry.name }) orelse continue; + const when = mtimeOf(io, path); + if (when <= best) continue; + if (path.len > out.len) continue; + @memcpy(out[0..path.len], path); + best = when; + best_len = path.len; + } + return if (best_len == 0) null else out[0..best_len]; +} + +/// The route that needs no knowledge of how a harness names its store: +/// whatever session file the process is holding open. The newest wins, +/// so a harness that keeps an old transcript open for reading does not +/// outvote the one it is writing. +fn resolveByFd(l: *Live, s: Sources, pid_dir: []const u8) void { + var root_buf: [path_capacity]u8 = undefined; + const root = sessionRoot(s, l.kind, &root_buf) orelse return; + + var fd_buf: [path_capacity]u8 = undefined; + const fd_dir_path = join(&fd_buf, &.{ pid_dir, "fd" }) orelse return; + const fd_dir = Io.Dir.openDirAbsolute(s.io, fd_dir_path, .{ .iterate = true }) catch return; + defer Io.Dir.close(fd_dir, s.io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(fd_dir, &rb); + + var best: i64 = -1; + var best_path: [path_capacity]u8 = undefined; + var best_len: usize = 0; + while (true) { + const entry = (it.next(s.io) catch break) orelse break; + var link_path_buf: [path_capacity]u8 = undefined; + const link_path = join(&link_path_buf, &.{ fd_dir_path, entry.name }) orelse continue; + var target_buf: [path_capacity]u8 = undefined; + const target = linkOf(s.io, link_path, &target_buf) orelse continue; + if (!std.mem.startsWith(u8, target, root)) continue; + if (target.len <= root.len or target[root.len] != '/') continue; + if (!sessionFile(l.kind, target[root.len + 1 ..])) continue; + const when = mtimeOf(s.io, target); + if (when <= best or target.len > best_path.len) continue; + @memcpy(best_path[0..target.len], target); + best = when; + best_len = target.len; + } + if (best_len == 0) return; + const found = best_path[0..best_len]; + l.transcript.set(found); + l.via = .fd; + if (l.kind == .codex) l.session.set(sessionIdOf(.codex, basename(found))); + if (l.kind == .dsh) { + // ~/.dsh/sessions//session-/session.jsonl.zstd: + // the id is the directory the transcript sits in. + const rel = found[root.len + 1 ..]; + var parts = std.mem.splitScalar(u8, rel, '/'); + _ = parts.next(); + if (parts.next()) |id| l.session.set(id); + } +} + +/// Is this path, relative to the harness's session root, one of its +/// session transcripts rather than a subagent's or something else? +fn sessionFile(kind: Kind, rel: []const u8) bool { + var depth: usize = 0; + var parts = std.mem.splitScalar(u8, rel, '/'); + while (parts.next()) |_| depth += 1; + return switch (kind) { + // /.jsonl exactly: a third component is a subagent. + .omp => depth == 2 and std.mem.endsWith(u8, rel, ".jsonl"), + .claude => depth == 2 and std.mem.endsWith(u8, rel, ".jsonl"), + .dsh => depth == 3 and std.mem.startsWith(u8, rel, "-"), + // YYYY/MM/DD/rollout--.jsonl + .codex => depth == 4 and std.mem.endsWith(u8, rel, ".jsonl") and + std.mem.startsWith(u8, basename(rel), "rollout-"), + .hermes => false, + }; +} + + +// ---- fields derived when they are read --------------------------------------- + +/// The value of `key` in a NUL-separated environment block, as +/// `/proc//environ` stores it. +pub fn envValue(environ: []const u8, key: []const u8) ?[]const u8 { + var it: Argv = .init(environ); + while (it.next()) |pair| { + if (pair.len <= key.len) continue; + if (!std.mem.startsWith(u8, pair, key)) continue; + if (pair[key.len] != '=') continue; + return pair[key.len + 1 ..]; + } + return null; +} + +/// The title key each harness writes into the head of a transcript. +fn titleKey(k: Kind) []const u8 { + return switch (k) { + .claude => "aiTitle", + else => "title", + }; +} + +/// A field from the first few records of a transcript, where the +/// harnesses put the session's title and the model it runs. Only the +/// head is read: these records are written when the session opens. +pub fn headField(io: Io, kind: Kind, transcript: []const u8, which: enum { title, model }, out: []u8) ?[]const u8 { + if (transcript.len == 0) return null; + var head: [probe_capacity]u8 = undefined; + const text = readSmall(io, transcript, &head) orelse return null; + const key = switch (which) { + .title => titleKey(kind), + .model => "model", + }; + const v = jsonField(text, key) orelse return null; + if (v.len == 0 or v.len > out.len) return null; + @memcpy(out[0..v.len], v); + return out[0..v.len]; +} + +/// The zmx session a process runs inside, from its environment. +pub fn zmxOf(s: Sources, pid: u32, out: []u8) ?[]const u8 { + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch return null; + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ s.proc, pid_text, "environ" }) orelse return null; + var env_buf: [probe_capacity]u8 = undefined; + const env = readSmall(s.io, path, &env_buf) orelse return null; + const v = envValue(env, "ZMX_SESSION") orelse return null; + if (v.len == 0 or v.len > out.len) return null; + @memcpy(out[0..v.len], v); + return out[0..v.len]; +} + +/// A field of Claude Code's own session record, which is the only +/// harness that publishes a name and a status. +pub fn registryField(s: Sources, l: *const Live, key: []const u8, out: []u8) ?[]const u8 { + if (l.kind != .claude) return null; + const base = rootOf(s, .claude) orelse return null; + var num: [24]u8 = undefined; + const file = std.fmt.bufPrint(&num, "{d}.json", .{l.pid}) catch return null; + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ base, "sessions", file }) orelse return null; + var text_buf: [probe_capacity]u8 = undefined; + const text = readSmall(s.io, path, &text_buf) orelse return null; + // The record must still describe this process, not a reused pid. + const proc_start = jsonField(text, "procStart") orelse return null; + const claimed = std.fmt.parseInt(u64, proc_start, 10) catch return null; + if (claimed != l.starttime) return null; + const v = jsonField(text, key) orelse return null; + if (v.len == 0 or v.len > out.len) return null; + @memcpy(out[0..v.len], v); + return out[0..v.len]; +} + +/// Is this process still the one the slot remembers? A pid alone is not +/// an identity: the kernel reuses them. +pub fn stillAlive(s: Sources, l: *const Live) bool { + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{l.pid}) catch return false; + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ s.proc, pid_text, "stat" }) orelse return false; + var stat_buf: [probe_capacity]u8 = undefined; + const stat = readSmall(s.io, path, &stat_buf) orelse return false; + const now = startTimeOf(stat) orelse return false; + return now == l.starttime; +} + +/// Seconds between the epoch and boot, from `/proc/stat`'s `btime` +/// line. Zero when it cannot be read. +pub fn bootTime(io: Io, proc: []const u8) i64 { + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ proc, "stat" }) orelse return 0; + var buf: [probe_capacity]u8 = undefined; + const text = readSmall(io, path, &buf) orelse return 0; + var lines = std.mem.splitScalar(u8, text, '\n'); + while (lines.next()) |line| { + if (!std.mem.startsWith(u8, line, "btime ")) continue; + return std.fmt.parseInt(i64, std.mem.trim(u8, line["btime ".len..], " \r"), 10) catch 0; + } + return 0; +} + + +// ---- subagents ---------------------------------------------------------------- + +/// Longest subagent name answered. +pub const agent_name_capacity: usize = 64; + +/// The subagents of one session: omp writes each as `.jsonl` in a +/// directory named after the session file. Sorted, so a listing that +/// spans several reads keeps its cursor. +pub const Agents = struct { + names: [max_agents][agent_name_capacity]u8 = undefined, + lens: [max_agents]u8 = @splat(0), + count: usize = 0, + + pub fn name(a: *const Agents, i: usize) []const u8 { + if (i >= a.count) return ""; + return a.names[i][0..a.lens[i]]; + } + + pub fn indexOf(a: *const Agents, want: []const u8) ?usize { + for (0..a.count) |i| if (std.mem.eql(u8, a.name(i), want)) return i; + return null; + } +}; + +pub fn agentsOf(io: Io, l: *const Live, out: *Agents) void { + out.count = 0; + const dir_path = l.agent_dir.slice(); + if (dir_path.len == 0) return; + const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch return; + defer Io.Dir.close(dir, io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(dir, &rb); + while (out.count < max_agents) { + const entry = (it.next(io) catch break) orelse break; + if (!std.mem.endsWith(u8, entry.name, ".jsonl")) continue; + const stem = entry.name[0 .. entry.name.len - ".jsonl".len]; + if (stem.len == 0 or stem.len > agent_name_capacity) continue; + @memcpy(out.names[out.count][0..stem.len], stem); + out.lens[out.count] = @intCast(stem.len); + out.count += 1; + } + // Insertion sort: the list is at most `max_agents` long and already + // nearly ordered, and this keeps the struct allocation-free. + var i: usize = 1; + while (i < out.count) : (i += 1) { + var j = i; + while (j > 0 and std.mem.order(u8, out.name(j), out.name(j - 1)) == .lt) : (j -= 1) { + std.mem.swap([agent_name_capacity]u8, &out.names[j], &out.names[j - 1]); + std.mem.swap(u8, &out.lens[j], &out.lens[j - 1]); + } + } +} + +/// The transcript path of subagent `i`, written into `out`. +pub fn agentPath(l: *const Live, a: *const Agents, i: usize, out: []u8) ?[]const u8 { + if (i >= a.count) return null; + var name_buf: [agent_name_capacity + 8]u8 = undefined; + const file = std.fmt.bufPrint(&name_buf, "{s}.jsonl", .{a.name(i)}) catch return null; + return join(out, &.{ l.agent_dir.slice(), file }); +} + +// ---- unit tests --------------------------------------------------------------- + +const testing = std.testing; + +/// A `/proc//cmdline`: NUL-separated argv, written out literally so +/// the tests read the way the kernel stores it. +fn cmd(comptime words: []const []const u8) []const u8 { + comptime var out: []const u8 = ""; + inline for (words) |w| out = out ++ w ++ "\x00"; + return out; +} + +test "active: a command line names its harness, or nothing" { + try testing.expectEqual(Kind.claude, harnessOf(cmd(&.{ "claude", "--dangerously-skip-permissions" })).?); + try testing.expectEqual(Kind.omp, harnessOf(cmd(&.{"omp"})).?); + try testing.expectEqual(Kind.codex, harnessOf(cmd(&.{ "/usr/bin/codex", "resume" })).?); + try testing.expectEqual(Kind.dsh, harnessOf(cmd(&.{"/home/goblin/.local/bin/dsh"})).?); + // hermes runs as a python module, named by a later word. + try testing.expectEqual(Kind.hermes, harnessOf(cmd(&.{ "/venv/bin/python", "-m", "hermes_cli.main" })).?); + // Not a harness at all. + try testing.expect(harnessOf(cmd(&.{ "bash", "-c", "claude" })) == null); + try testing.expect(harnessOf("") == null); + try testing.expect(harnessOf("\x00\x00") == null); +} + +test "active: a daemon or helper wearing a harness name is never a session" { + // The live machine's own gateway and dashboard, which must not list. + try testing.expect(harnessOf(cmd(&.{ "python3", "-m", "hermes_cli.main", "gateway", "run" })) == null); + try testing.expect(harnessOf(cmd(&.{ "python3", "/x/hermes-dashboard", "dashboard" })) == null); + try testing.expect(harnessOf(cmd(&.{ "claude", "mcp-server" })) == null); + try testing.expect(harnessOf(cmd(&.{ "omp", "__omp_worker_daemon_broker" })) == null); + try testing.expect(harnessOf(cmd(&.{ "codex", "app-server" })) == null); + // A directory that merely contains the word is still a session: the + // exclusion matches a whole argument, never a substring. + try testing.expectEqual(Kind.claude, harnessOf(cmd(&.{ "claude", "--cwd", "/home/goblin/serve-me" })).?); +} + +test "active: starttime is counted from the last ')' in a stat line" { + // Fields: pid (comm) state ppid ... starttime is field 22. + var line: [512]u8 = undefined; + var w = Io.Writer.fixed(&line); + try w.print("345104 (claude) S 344980", .{}); + for (5..22) |i| try w.print(" {d}", .{i * 10}); + try w.print(" 1625063 rest of the line", .{}); + try testing.expectEqual(@as(u64, 1625063), startTimeOf(w.buffered()).?); + + // A process whose name holds spaces and parentheses — the reason + // the fields are never counted from the start of the line. + var nasty: [512]u8 = undefined; + var nw = Io.Writer.fixed(&nasty); + try nw.print("42 (we i)rd (name) ) R 1", .{}); + for (5..22) |i| try nw.print(" {d}", .{i}); + try nw.print(" 999 x", .{}); + try testing.expectEqual(@as(u64, 999), startTimeOf(nw.buffered()).?); + + try testing.expect(startTimeOf("no parens here") == null); + try testing.expect(startTimeOf("1 (short) S 2 3") == null); +} + +test "active: json fields come out of a harness's own session record" { + // The shape Claude Code writes to ~/.claude/sessions/.json. + const rec = + \\{ + \\ "pid": 345104, + \\ "sessionId": "1f9a74f7-b04f-4092-8b98-df11316e03c0", + \\ "cwd": "/home/goblin", + \\ "version": "2.1.278", + \\ "peerFeatures": ["notify_idle", "artifact_yield"], + \\ "name": "goblin-e5", + \\ "status": "busy", + \\ "procStart": "1625063" + \\} + ; + try testing.expectEqualStrings("345104", jsonField(rec, "pid").?); + try testing.expectEqualStrings("1f9a74f7-b04f-4092-8b98-df11316e03c0", jsonField(rec, "sessionId").?); + try testing.expectEqualStrings("/home/goblin", jsonField(rec, "cwd").?); + try testing.expectEqualStrings("goblin-e5", jsonField(rec, "name").?); + try testing.expectEqualStrings("busy", jsonField(rec, "status").?); + try testing.expectEqualStrings("1625063", jsonField(rec, "procStart").?); + // An array or object value is skipped, not half-parsed. + try testing.expect(jsonField(rec, "peerFeatures") == null); + try testing.expect(jsonField(rec, "absent") == null); + // A key that is a prefix of another must not match it. + try testing.expect(jsonField("{\"sessionIdent\":\"x\"}", "sessionId") == null); +} + +test "active: session ids and store slugs follow each harness's spelling" { + try testing.expectEqualStrings( + "01a0c46a-7a06-7676-aef1-add9d9340c83", + sessionIdOf(.omp, "2026-09-21T14-41-47-526Z_01a0c46a-7a06-7676-aef1-add9d9340c83.jsonl"), + ); + try testing.expectEqualStrings( + "42898558-2fa0-41f2-a2c6-1357e5cf6836", + sessionIdOf(.claude, "42898558-2fa0-41f2-a2c6-1357e5cf6836.jsonl"), + ); + // codex's timestamp holds dashes too, so the id is the last 36 + // bytes and never what splitting on `-` would give. + try testing.expectEqualStrings( + "01a0bcd2-5653-7cc0-aa36-676f6467a72a", + sessionIdOf(.codex, "rollout-2026-09-20T00-18-16-01a0bcd2-5653-7cc0-aa36-676f6467a72a.jsonl"), + ); + + var buf: [256]u8 = undefined; + // omp strips $HOME and keeps everything but the slashes. + try testing.expectEqualStrings( + "-00-projects-0x4200.cafe-cloud9", + slugOf(.omp, "/home/goblin/00-projects/0x4200.cafe/cloud9", "/home/goblin", &buf).?, + ); + // Claude Code dashes every byte that is not a letter or a digit. + try testing.expectEqualStrings( + "-home-goblin-00-projects-0x4200-cafe-cloud9", + slugOf(.claude, "/home/goblin/00-projects/0x4200.cafe/cloud9", "/home/goblin", &buf).?, + ); + // A path longer than the buffer is refused, never truncated into a + // slug that would name the wrong store. + var tiny: [4]u8 = undefined; + try testing.expect(slugOf(.omp, "/home/goblin/somewhere", "/home/goblin", &tiny) == null); +} + +test "active: a zmx session name is checked before it is ever used" { + try testing.expect(legalZmxName("harness")); + try testing.expect(legalZmxName("claude-cloud9.2_x")); + try testing.expect(!legalZmxName("")); + try testing.expect(!legalZmxName("has space")); + try testing.expect(!legalZmxName("../escape")); + try testing.expect(!legalZmxName("semi;colon")); + try testing.expect(!legalZmxName("nul\x00byte")); + try testing.expect(!legalZmxName("x" ** 65)); +} + +test "active: an environment block answers by key, not by prefix" { + const env = "PATH=/usr/bin\x00ZMX_SESSION=harness\x00HOME=/home/goblin\x00"; + try testing.expectEqualStrings("harness", envValue(env, "ZMX_SESSION").?); + try testing.expectEqualStrings("/home/goblin", envValue(env, "HOME").?); + try testing.expect(envValue(env, "ZMX") == null); // a prefix is not the key + try testing.expect(envValue(env, "NOPE") == null); + try testing.expect(envValue("", "HOME") == null); + // An entry with no value is a value of zero length, not a miss. + try testing.expectEqualStrings("", envValue("EMPTY=\x00", "EMPTY").?); +} diff --git a/9agents/src/main.zig b/9agents/src/main.zig new file mode 100644 index 0000000..2ac9078 --- /dev/null +++ b/9agents/src/main.zig @@ -0,0 +1,382 @@ +//! 9agents: the coding-agents fs daemon — a long-running 9P2000 server that +//! mirrors every AI-agent harness's state (Claude Code, Codex, omp, +//! hermes, dsh + their skills) as one read-only, fresh-from-disk tree. +//! See docs/DESIGN.md for the tree contract and src/tree.zig for the tree. +//! +//! By default it posts itself under the name `agents` with +//! `serve.Runner.listenPosted`, so it appears at $XDG_RUNTIME_DIR/9p/agents +//! and every interactive fish (self-wrapped in `9ns --mntgen`) sees it at +//! /mnt/9p/agents with zero configuration. `--unix`, `--tcp` and `--fd` +//! are the other listen forms; `--no-post` serves without posting; the +//! five harness roots default to $HOME/. and `--root NAME=PATH` +//! pins any of them elsewhere (the tests use fake homes). +//! +//! v1 has no daemonization: it never forks or detaches, so it is run under +//! something that supervises it — a systemd user service with +//! `Restart=always` (see 9agents.service), or a zmx session: +//! +//! zmx run agents -d 9agents +//! +//! SIGTERM or SIGINT stops it cleanly (a posted name is unposted). +const std = @import("std"); +const builtin = @import("builtin"); +const cloud9 = @import("cloud9"); +const tree = @import("tree.zig"); + +const fs = cloud9.fs; +const serve = cloud9.serve; +const post = cloud9.post; +const transport = cloud9.transport; +const Io = std.Io; +const linux = std.os.linux; + +const usage_text = + \\usage: 9agents [--unix PATH | --tcp IP:PORT | --fd N] [--no-post] + \\ [--name NAME] [--root NAME=PATH]... [--proc DIR] + \\ [--allow-move] [--zmx PATH] + \\ + \\A read-only, fresh-from-disk 9P2000 view of every harness's state: + \\ /README the tree, explained in place + \\ /pid /uptime daemon facts + \\ /claude/{projects,history,skills} + \\ /codex/{sessions,session-index,history} + \\ /omp /hermes /dsh full mirrors, raw + \\ /skills/{claude,codex,omp} the union skills view + \\ /active/// what is running right now + \\ + \\By default the daemon posts itself under the name `agents`, so it is + \\dialable at $XDG_RUNTIME_DIR/9p/agents and mountable by 9ns --mntgen + \\at /mnt/9p/agents. --no-post skips posting; --unix/--tcp add plain + \\listeners beside the post; --fd N serves one 9P session over the + \\connected stream on descriptor N and posts nothing. + \\--root NAME=PATH pins one harness root (NAME: claude, codex, omp, + \\hermes, dsh) somewhere other than $HOME/.; repeatable. + \\--proc DIR reads the process tree somewhere other than /proc (for + \\tests); --proc "" leaves /active out of the tree entirely. + \\--allow-move lets a write to /active///zmx move that agent + \\into a zmx session of that name: the daemon kills it and re-execs + \\the harness under zmx with its session resumed. Off by default, + \\because it is the one place the tree is not read-only, and anything + \\that can mount it can then kill an agent. --zmx PATH names the + \\binary a move runs (default: zmx, found on $PATH). + \\ + \\Run it in a zmx session, the zmx way: `zmx run agents -d 9agents`. + \\ +; + +const max_connections = 16; +const Runner = serve.Runner(tree.Harness, tree.opts, .{ + .msize = tree.msize, + .connections = max_connections, + .listeners = 2, // the posted name plus one of --unix/--tcp +}); + +// Static memory, the 9proc-demo way: the harness and the runner are large +// (the path table, the per-connection buffers) and live in .bss. +var harness_mem: tree.Harness = undefined; +var runner_mem: Runner = undefined; +var stop_requested = std.atomic.Value(bool).init(false); + +fn onSignal(sig: linux.SIG) callconv(.c) void { + _ = sig; + stop_requested.store(true, .release); +} + +fn installSignalHandlers() void { + const act = linux.Sigaction{ + .handler = .{ .handler = onSignal }, + .mask = @splat(0), + .flags = 0, + }; + _ = linux.sigaction(.TERM, &act, null); + _ = linux.sigaction(.INT, &act, null); + const ign = linux.Sigaction{ + .handler = .{ .handler = linux.SIG.IGN }, + .mask = @splat(0), + .flags = 0, + }; + _ = linux.sigaction(.PIPE, &ign, null); +} + +// ---- the serve handler --------------------------------------------------------- + +fn serveReq(ctx: ?*anyopaque, conn: *Runner.Conn, req: fs.Req) void { + const h: *tree.Harness = @ptrCast(@alignCast(ctx.?)); + // The answer's bytes point into the harness's shared buffers, so the + // mutex spans handle and reply: the engine copies them into the + // connection's output before the lock goes. + h.mutex.lockUncancelable(h.io); + const a = tree.handle(h, req); + conn.reply(&a.reply, a.bytes); + h.mutex.unlock(h.io); +} + +/// `name` found on `path_env`, as an absolute path in the arena. +fn onPath(io: Io, arena: std.mem.Allocator, path_env: []const u8, name: []const u8) ![]const u8 { + var it = std.mem.splitScalar(u8, path_env, ':'); + while (it.next()) |dir_path| { + if (dir_path.len == 0) continue; + const candidate = try std.fmt.allocPrint(arena, "{s}/{s}", .{ dir_path, name }); + const st = Io.Dir.statFile(.cwd(), io, candidate, .{ .follow_symlinks = true }) catch continue; + if (st.kind != .file) continue; + return candidate; + } + return error.NotFound; +} + +// ---- the CLI -------------------------------------------------------------------- + +const Mode = enum { posted, unix, tcp, fd }; + +pub fn main(init: std.process.Init) !void { + run(init) catch |e| switch (e) { + // Both are reported with their own line above; a returned error + // would bury it under a Debug-build stack trace. + error.Usage, error.AlreadyPosted => std.process.exit(1), + else => return e, + }; +} + +fn run(init: std.process.Init) !void { + const io = init.io; + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + const envp: post.Env = init.minimal.environ.block.slice.ptr; + + var mode: Mode = .posted; + var unix_path: []const u8 = ""; + var tcp_addr: []const u8 = ""; + var fd_no: i32 = -1; + var post_name: []const u8 = "agents"; + var no_post = false; + var base_overrides: [5]?[]const u8 = @splat(null); + var proc_root: []const u8 = "/proc"; + var zmx_path: []const u8 = "zmx"; + var allow_move = false; + + var i: usize = 1; + while (i < args.len) : (i += 1) { + const a = args[i]; + if (std.mem.eql(u8, a, "--no-post")) { + no_post = true; + } else if (std.mem.eql(u8, a, "--allow-move")) { + allow_move = true; + } else if (std.mem.eql(u8, a, "--unix") or std.mem.eql(u8, a, "--tcp") or + std.mem.eql(u8, a, "--fd") or std.mem.eql(u8, a, "--name") or + std.mem.eql(u8, a, "--proc") or std.mem.eql(u8, a, "--zmx") or + std.mem.eql(u8, a, "--root")) + { + i += 1; + if (i >= args.len) { + std.debug.print("9agents: {s} needs an argument\n{s}", .{ a, usage_text }); + return error.Usage; + } + const v = args[i]; + if (std.mem.eql(u8, a, "--unix")) { + mode = .unix; + unix_path = v; + } else if (std.mem.eql(u8, a, "--tcp")) { + mode = .tcp; + tcp_addr = v; + } else if (std.mem.eql(u8, a, "--fd")) { + mode = .fd; + fd_no = std.fmt.parseInt(i32, v, 10) catch { + std.debug.print("9agents: --fd: not a number: {s}\n", .{v}); + return error.Usage; + }; + } else if (std.mem.eql(u8, a, "--proc")) { + proc_root = v; + } else if (std.mem.eql(u8, a, "--zmx")) { + zmx_path = v; + } else if (std.mem.eql(u8, a, "--name")) { + post_name = v; + if (!post.legalName(post_name)) { + std.debug.print("9agents: illegal post name: {s}\n", .{v}); + return error.Usage; + } + } else { + const eq = std.mem.indexOfScalar(u8, v, '=') orelse { + std.debug.print("9agents: --root NAME=PATH: {s}\n{s}", .{ v, usage_text }); + return error.Usage; + }; + const name = v[0..eq]; + const root = std.meta.stringToEnum(tree.Root, name) orelse { + std.debug.print("9agents: unknown root {s} (claude, codex, omp, hermes, dsh)\n", .{name}); + return error.Usage; + }; + base_overrides[@intFromEnum(root)] = v[eq + 1 ..]; + } + } else if (std.mem.eql(u8, a, "--help") or std.mem.eql(u8, a, "-h")) { + std.debug.print("{s}", .{usage_text}); + return; + } else { + std.debug.print("9agents: unknown argument {s}\n{s}", .{ a, usage_text }); + return error.Usage; + } + } + + // The five pinned roots: $HOME/. unless overridden. + const home = post.getenv(envp, "HOME"); + var bases: [5][]const u8 = @splat(""); + inline for (0..5) |r| { + if (base_overrides[r]) |path| { + bases[r] = path; + } else if (home) |hm| { + bases[r] = std.fmt.bufPrint(&root_bufs[r], "{s}/.{s}", .{ hm, root_dirs[r] }) catch return error.NameTooLong; + } + } + var any = false; + for (bases) |b| any = any or b.len != 0; + if (!any) { + std.debug.print("9agents: no harness roots (set $HOME or pass --root NAME=PATH)\n", .{}); + return error.Usage; + } + + // A move execs zmx by absolute path: the daemon resolves it once, + // at startup, so nothing about $PATH matters when the write lands. + const zmx_abs = if (std.mem.indexOfScalar(u8, zmx_path, '/') != null) + zmx_path + else + onPath(io, arena, post.getenv(envp, "PATH") orelse "", zmx_path) catch zmx_path; + if (allow_move and std.mem.indexOfScalar(u8, zmx_abs, '/') == null) { + std.debug.print("9agents: --allow-move needs zmx on $PATH (or --zmx PATH)\n", .{}); + return error.Usage; + } + + harness_mem.init(.{ + .io = io, + .pid = @intCast(linux.getpid()), + .bases = bases, + .proc = proc_root, + .home = home orelse "", + .zmx = zmx_abs, + .runtime = post.getenv(envp, "XDG_RUNTIME_DIR") orelse "", + .allow_move = allow_move, + .envp = envp, + }); + if (allow_move) { + std.debug.print("9agents: moves allowed — a write to /active///zmx re-execs that agent under {s}\n", .{zmx_abs}); + } + if (no_post and mode == .posted) { + std.debug.print("9agents: --no-post needs a listen form (--unix, --tcp or --fd)\n{s}", .{usage_text}); + return error.Usage; + } + + installSignalHandlers(); + + switch (mode) { + .fd => { + std.debug.print("9agents: serving one session on fd {d}\n", .{fd_no}); + try serveFd(io, fd_no); + return; + }, + .posted, .unix, .tcp => {}, + } + + runner_mem.init(.{ .io = io, .root = tree.root, .handler = .{ .ctx = &harness_mem, .serve = serveReq } }); + defer runner_mem.stop(); + + var posted_something = false; + if (!no_post) { + runner_mem.listenPosted(envp, post_name, 16) catch |err| { + if (err == error.AlreadyPosted) { + std.debug.print("9agents: the name `{s}` is already posted by a live server\n", .{post_name}); + return err; + } + std.debug.print("9agents: cannot post as {s}: {t}\n", .{ post_name, err }); + return err; + }; + posted_something = true; + var pbuf: [transport.sun_path_len]u8 = undefined; + const path = post.registryPath(envp, post_name, &pbuf) catch ""; + std.debug.print("9agents: posted as {s} at {s}\n", .{ post_name, path }); + } + switch (mode) { + .unix => { + _ = try runner_mem.listen(.{ .unix = try arena.dupeZ(u8, unix_path) }, 16); + std.debug.print("9agents: listening on {s}\n", .{unix_path}); + }, + .tcp => { + const addr = Io.net.IpAddress.parseLiteral(tcp_addr) catch { + std.debug.print("9agents: bad --tcp address: {s}\n", .{tcp_addr}); + return error.Usage; + }; + const bound = try runner_mem.listen(.{ .tcp = addr }, 16); + std.debug.print("9agents: listening on tcp!{f}\n", .{bound}); + }, + else => {}, + } + var pinned: u32 = 0; + for (bases) |b| { + if (b.len != 0) pinned += 1; + } + std.debug.print("9agents: serving ({d} roots pinned, {d} bytes of tables)\n", .{ pinned, @sizeOf(tree.Harness) }); + + // The runner's tasks drive themselves; the main thread only waits for + // a stop signal (SIGTERM/SIGINT) and then unposts via stop(). + while (!stop_requested.load(.acquire)) { + io.sleep(.fromMilliseconds(250), .awake) catch break; + } + std.debug.print("9agents: stopping\n", .{}); +} + +const root_dirs = [5][]const u8{ "claude", "codex", "omp", "hermes", "dsh" }; +var root_bufs: [5][tree.base_capacity]u8 = @splat(@splat(0)); + +// ---- --fd N: one 9P session over a connected stream ----------------------------- + +/// Serves exactly one client on the connected stream of descriptor `n` +/// (socket activation, `9ns --spawn`-style handoffs), driving the engine +/// directly, the freestanding way. Returns when the client hangs up. +fn serveFd(io: Io, n: i32) !void { + const stream: Io.net.Stream = .{ .socket = .{ .handle = n, .address = .{ .ip4 = .loopback(0) } } }; + const Engine = fs.Server(tree.Harness, tree.opts); + var in: [tree.msize]u8 = undefined; + var out: [2 * tree.msize]u8 = undefined; + var rbuf: [tree.msize]u8 = undefined; + var wbuf: [2 * tree.msize]u8 = undefined; + var stage: [tree.msize]u8 = undefined; + var engine = Engine.init(.{ .in = &in, .out = &out, .root = tree.root }); + var reader = stream.reader(io, &rbuf); + var writer = stream.writer(io, &wbuf); + while (true) { + const frame = transport.readFrame(&reader.interface, &stage, tree.msize) catch return; + var off: usize = 0; + while (off < frame.len) { + off += engine.push(frame[off..]); + while (true) { + const req = engine.next() orelse break; + const a = answer(req); + engine.reply(&a.reply, a.bytes); + } + const pending = engine.output(); + writer.interface.writeAll(pending) catch return; + engine.wrote(pending.len); + if (engine.protocol.dead) return; + } + writer.interface.flush() catch return; + } +} + +/// One request for the --fd loop: same locking discipline as the runner's +/// handler (the reply bytes live in the harness's shared buffers). +fn answer(req: fs.Req) tree.Answer { + harness_mem.mutex.lockUncancelable(harness_mem.io); + defer harness_mem.mutex.unlock(harness_mem.io); + return tree.handle(&harness_mem, req); +} + +test { + _ = @import("tree.zig"); // the tree's unit tests (exclusions, ids, ...) +} + +test "main: the root override parser pins each named root" { + // Parsing is inline in run(); the mapping itself is what the tests + // rely on, and it is exercised end to end in test/e2e.sh. + _ = std.meta.stringToEnum(tree.Root, "claude").?; + _ = std.meta.stringToEnum(tree.Root, "codex").?; + _ = std.meta.stringToEnum(tree.Root, "omp").?; + _ = std.meta.stringToEnum(tree.Root, "hermes").?; + _ = std.meta.stringToEnum(tree.Root, "dsh").?; + try std.testing.expect(std.meta.stringToEnum(tree.Root, "zmx") == null); +} diff --git a/9agents/src/tree.zig b/9agents/src/tree.zig new file mode 100644 index 0000000..ed0699c --- /dev/null +++ b/9agents/src/tree.zig @@ -0,0 +1,2413 @@ +//! The harness file tree served over 9P2000: a unified, read-only, +//! fresh-from-disk view of every AI-agent harness's state on the machine. +//! +//! /README what the tree is and how to read it +//! /pid /uptime daemon facts, one line each +//! /claude/ projects/ (mirror of ~/.claude/projects), +//! history (~/.claude/history.jsonl), +//! skills/ (mirror of ~/.claude/skills) +//! /codex/ sessions/ (~/.codex/sessions), +//! session-index (~/.codex/session_index.jsonl), +//! history (~/.codex/history.jsonl) +//! /omp/ mirror of ~/.omp/agent, raw blobs +//! /hermes/ mirror of ~/.hermes, raw +//! /dsh/ mirror of ~/.dsh +//! /skills/{claude,codex,omp}/ the same skills trees the harnesses own +//! +//! Everything below the named mount points is a lazy mirror: a lookup, +//! getattr or readdir walks the real filesystem at request time, so a +//! session transcript grows as its harness writes it and a new session +//! appears as soon as its file lands. No cache, no invalidation. +//! +//! Security boundary — the exclusion rule is absolute and unit-tested: +//! no name that looks like a credential, key, token or auth store is ever +//! answered, at any depth (`excluded`); symlinks are never served (they +//! are an escape hatch around the pinned roots). Every path is resolved +//! from its pinned root one component at a time with `O_NOFOLLOW` +//! (`openIn`), so no name below a root — swapped mid-session or not — +//! can point the daemon at a file outside it. The daemon must never +//! become a credential reader for anything that mounts it. Writes, +//! creates and setattrs answer EPERM. +//! +//! This module is the backend of `cloud9.fs.Server` (main.zig hands it to +//! `serve.Runner`). It declares `features = .{ .references = true }`: every +//! lookup result is a reference, so table entries below (the mirrored +//! files) are refcounted and freed when the last fid lets go. Node ids +//! pack (kind, root index, serial) in a u64, zmx-style: a mirrored file's +//! serial is its table slot and generation, so two walks of the same file +//! answer the same qid path while it is held. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fs = cloud9.fs; +const E = fs.E; +const Io = std.Io; +const linux = std.os.linux; +pub const active = @import("active.zig"); + +// ---- comptime bounds --------------------------------------------------------- + +/// Longest relative path a mirrored file may carry (bytes below a root). +pub const rel_capacity: usize = 640; +/// Mirrored files remembered at once; each holds one reference per fid. +pub const path_capacity: usize = 4096; +comptime { + if (path_capacity > 1 << 12) @compileError("path table slots must fit the 12 serial bits"); +} +/// Largest frame the daemon negotiates; read and readdir staging buffers +/// are this big, so a single Rread can carry one full frame of bytes. +pub const msize: u32 = 32 * 1024; +/// Names one directory listing may stage (sorted for cursor stability). +pub const list_capacity: usize = 1024; +/// Bytes of name storage one listing may use (255 per name, packed). +pub const list_name_bytes: usize = 128 * 1024; +/// Longest base path a pinned root may carry. +pub const base_capacity: usize = 512; + +pub const opts: fs.Options = .{ + .fid_capacity = 512, + .slot_capacity = 8, + .name_capacity = 255, + .username_capacity = 28, + .fid_index = true, +}; + +// ---- the tree skeleton ------------------------------------------------------ + +pub const Root = enum(u8) { claude, codex, omp, hermes, dsh }; + +/// Static nodes: the facts, the harness dirs and the union skills dir. +pub const Top = enum(u8) { + root = 1, + pid, + uptime, + claude, + codex, + omp, + hermes, + dsh, + skills, + active, + readme, + + /// Directory entries of the root, in listing order. + pub const listed = [_]Top{ .readme, .pid, .uptime, .claude, .codex, .omp, .hermes, .dsh, .skills, .active }; + + pub fn fileName(t: Top) []const u8 { + // The only name that is not its tag: a README announces itself in + // the listing, and lowercase would bury it among the harness dirs. + return if (t == .readme) "README" else @tagName(t); + } + + pub fn dir(t: Top) bool { + return switch (t) { + .root, .claude, .codex, .omp, .hermes, .dsh, .skills, .active => true, + .pid, .uptime, .readme => false, + }; + } + + fn parent(t: Top) Top { + _ = t; + return .root; // the root is its own parent; every top hangs off it + } + + /// The mirror a harness dir serves: its root and the relative path + /// below it. The claude and codex dirs are *virtual* (their children + /// are named mounts); omp, hermes and dsh mirror one subtree each. + fn mirror(t: Top) ?Mount { + return switch (t) { + .omp => .{ .root = .omp, .rel = "agent" }, + .hermes => .{ .root = .hermes, .rel = "" }, + .dsh => .{ .root = .dsh, .rel = "" }, + else => null, + }; + } +}; + +/// Where a mirrored subtree sits: `rel` below the root's pinned base path. +pub const Mount = struct { + root: Root, + rel: []const u8, +}; + +const NamedMount = struct { + /// The name as it appears in its virtual directory. + name: []const u8, + owner: Top, + mount: Mount, +}; + +/// The children of the virtual directories (claude, codex, skills). The +/// same target may appear twice (/claude/skills and /skills/claude): a +/// lookup of either hands out the same table entry, so both paths share +/// qid paths and identity. +const named_mounts = [_]NamedMount{ + .{ .name = "projects", .owner = .claude, .mount = .{ .root = .claude, .rel = "projects" } }, + .{ .name = "history", .owner = .claude, .mount = .{ .root = .claude, .rel = "history.jsonl" } }, + .{ .name = "skills", .owner = .claude, .mount = .{ .root = .claude, .rel = "skills" } }, + .{ .name = "sessions", .owner = .codex, .mount = .{ .root = .codex, .rel = "sessions" } }, + .{ .name = "session-index", .owner = .codex, .mount = .{ .root = .codex, .rel = "session_index.jsonl" } }, + .{ .name = "history", .owner = .codex, .mount = .{ .root = .codex, .rel = "history.jsonl" } }, + .{ .name = "claude", .owner = .skills, .mount = .{ .root = .claude, .rel = "skills" } }, + .{ .name = "codex", .owner = .skills, .mount = .{ .root = .codex, .rel = "skills" } }, + .{ .name = "omp", .owner = .skills, .mount = .{ .root = .omp, .rel = "skills" } }, +}; + +fn mountNamed(owner: Top, name: []const u8) ?Mount { + for (named_mounts) |m| { + if (m.owner == owner and std.mem.eql(u8, m.name, name)) return m.mount; + } + return null; +} + +/// The Top that owns the mount whose target is exactly (root, rel) — the +/// canonical parent a ".." walk answers. Falls back to the root's harness +/// dir for targets no named mount covers (deep paths, mirrors). +fn mountOwner(r: Root, rel_path: []const u8) Top { + for (named_mounts) |m| { + if (m.mount.root == r and std.mem.eql(u8, m.mount.rel, rel_path)) return m.owner; + } + return switch (r) { + .claude => .claude, + .codex => .codex, + .omp => .omp, + .hermes => .hermes, + .dsh => .dsh, + }; +} + +// ---- node ids and the path table --------------------------------------------- + +pub const Kind = enum(u8) { top = 0, path = 1, active = 2 }; + +comptime { + // The derived view indexes the same pinned roots the mirror does. + for (std.enums.values(Root)) |r| { + if (!std.mem.eql(u8, @tagName(r), @tagName(@as(active.Kind, @enumFromInt(@intFromEnum(r)))))) + @compileError("tree.Root and active.Kind must agree, index by index"); + } +} + +/// A node under `/active`. Everything but a harness directory names a +/// slot of the live table and the generation it had when the id was +/// minted, so an id outlives its process only long enough to answer +/// ENOENT. +pub const AFile = enum(u8) { + /// `/active/` + harness_dir, + /// `/active//` + entry, + pid, + ppid, + started, + cwd, + name, + title, + session, + model, + via, + zmx, + status, + transcript, + /// `/active///agents` + agents, + /// `/active///agents/` + agent, + agent_model, + agent_transcript, + + /// The fields of one entry, in listing order. A field with no value + /// is not listed and does not resolve. + pub const entry_files = [_]AFile{ + .pid, .ppid, .started, .cwd, .name, .title, + .session, .model, .via, .zmx, .status, .transcript, + }; + pub const agent_files = [_]AFile{ .agent_model, .agent_transcript }; + + pub fn fileName(f: AFile) []const u8 { + return switch (f) { + .agent_model => "model", + .agent_transcript => "transcript", + else => @tagName(f), + }; + } +}; + +/// A resolved `/active` node. +pub const Act = struct { + file: AFile, + /// Only for `harness_dir`. + kind: active.Kind = .claude, + slot: u8 = 0, + gen: u8 = 0, + agent: u8 = 0, +}; + +pub fn activeNode(a: Act) u64 { + const serial: u48 = switch (a.file) { + .harness_dir => @intFromEnum(a.kind), + else => @as(u48, a.slot) | (@as(u48, a.gen) << 8) | (@as(u48, a.agent) << 16), + }; + return @bitCast(Node{ + .idx = @intFromEnum(a.file), + .kind = @intFromEnum(Kind.active), + .serial = serial, + }); +} + +pub const Node = packed struct(u64) { + /// A Top index (kind == .top) or a Root index (kind == .path). + idx: u8 = 0, + kind: u8 = 0, + serial: u48 = 0, +}; + +pub fn topNode(t: Top) u64 { + return @bitCast(Node{ .idx = @intFromEnum(t), .kind = @intFromEnum(Kind.top) }); +} + +pub const root: u64 = topNode(Top.root); + +/// One remembered mirrored file: its root, its relative path, the hash +/// that short-circuits dedupe lookups, and its reference count (one per +/// fid holding it, handed out by lookup, paid back by release). +const Entry = struct { + used: bool = false, + root: Root = .claude, + rel_buf: [rel_capacity]u8 = undefined, + rel_len: u16 = 0, + hash: u64 = 0, + refs: u32 = 0, + gen: u32 = 0, + + fn rel(e: *const Entry) []const u8 { + return e.rel_buf[0..e.rel_len]; + } +}; + +fn serialOf(slot: u12, gen: u32) u48 { + return (@as(u48, gen) << 12) | slot; +} + +fn nodeOf(e: *const Entry, slot: u12) u64 { + return @bitCast(Node{ + .idx = @intFromEnum(e.root), + .kind = @intFromEnum(Kind.path), + .serial = serialOf(slot, e.gen), + }); +} + +// ---- the exclusions (the security boundary) ----------------------------------- + +fn containsFold(name: []const u8, needle: []const u8) bool { + if (name.len < needle.len) return false; + var i: usize = 0; + while (i + needle.len <= name.len) : (i += 1) { + if (std.ascii.eqlIgnoreCase(name[i .. i + needle.len], needle)) return true; + } + return false; +} + +fn endsWithFold(name: []const u8, suffix: []const u8) bool { + return name.len >= suffix.len and std.ascii.eqlIgnoreCase(name[name.len - suffix.len ..], suffix); +} + +fn isOneOf(name: []const u8, comptime names: []const []const u8) bool { + inline for (names) |n| if (std.ascii.eqlIgnoreCase(name, n)) return true; + return false; +} + +/// Absolute rule: a name that may hold a credential, key, token or auth +/// material is never served, at any depth, by lookup or readdir. The +/// substrings are folded (case-insensitive) and deliberately broad — +/// "auth" also hides "author-notes", the price of never guessing wrong. +/// When unsure, exclude and document (docs/DESIGN.md). +pub fn excluded(name: []const u8) bool { + @setEvalBranchQuota(20000); + if (name.len == 0) return true; + for ([_][]const u8{ "credentials", "token", "auth", "secret" }) |needle| { + if (containsFold(name, needle)) return true; + } + if (endsWithFold(name, ".key")) return true; + if (endsWithFold(name, ".pem")) return true; + if (endsWithFold(name, ".env")) return true; + // Config and settings files of the other harnesses may embed API keys + // (hermes config.yaml, omp config.yml, dsh settings.yaml). Claude + // Code's settings.json is not in the tree at all: /claude serves only + // projects/, skills/ and history.jsonl. + if (isOneOf(name, &.{ "settings.json", "settings.yaml", "settings.local.json", "config.yml", "config.yaml", "config.toml", "models.yml", "models.yaml", ".ssh" })) return true; + for ([_][]const u8{ "id_rsa", "id_dsa", "id_ecdsa", "id_ed25519" }) |prefix| { + if (name.len >= prefix.len and std.ascii.eqlIgnoreCase(name[0..prefix.len], prefix)) return true; + } + return false; +} + +/// Every component of a relative path must pass `excluded`; used on the +/// comptime mount targets and the unit-test fixtures alike. +pub fn excludedPath(rel_path: []const u8) bool { + @setEvalBranchQuota(4000); + var it = std.mem.splitScalar(u8, rel_path, '/'); + while (it.next()) |comp| { + if (excluded(comp)) return true; + } + return false; +} + +// ---- the backend -------------------------------------------------------------- + +pub const Answer = struct { + reply: fs.Reply, + bytes: []const u8 = "", +}; + +fn fail(tag: u64, e: u16) Answer { + return .{ .reply = fs.Reply.fail(tag, e) }; +} + +/// A refusal that says why. 9P2000 has no numeric error — error(5) is +/// `Rerror tag[2] ename[s]`, a string — and acme names its refusals in words +/// ("permission denied", "file does not exist", fsys.c). A server that only +/// answers bare errnos throws away the one channel the protocol gives it for +/// explaining itself, which matters most here: a move has seven different +/// reasons to say no, and EPERM tells the user none of them. +/// +/// `errno` is still carried, because the reader may be a Linux mount rather +/// than a person: 9ns maps the *text* back to an errno by case-insensitive +/// substring (enameToErrno in 9ns/src/nine.zig, first match wins). So each +/// message below is worded to keep the phrase that carries its errno — +/// "not permitted" for EPERM, "invalid" for EINVAL, "exists" for EEXIST, +/// "no such" for ENOENT — and to avoid phrases that would silently retarget +/// it ("permission" and "denied" map to EACCES, "read-only" to EROFS, and +/// the bare substring "fid" to EBADF). +fn failWhy(tag: u64, e: u16, ename: []const u8) Answer { + return .{ .reply = .{ .tag = tag, .status = .err, .errno = e, .ename = ename } }; +} + +/// The daemon's state. One instance serves every connection; `handle` is +/// called with `mutex` held (the serve handler in main.zig holds it across +/// handle and reply, because `bytes` point into the shared buffers). +pub const Harness = struct { + pub const Req = fs.Req; + pub const Reply = fs.Reply; + pub const features: fs.Features = .{ .references = true }; + + io: Io, + pid: u32 = 0, + started_sec: i64 = 0, + /// The five pinned roots; a root without a base is unreachable. + base_buf: [5][base_capacity]u8 = @splat(@splat(0)), + base_len: [5]u16 = @splat(0), + base_set: [5]bool = @splat(false), + table: [path_capacity]Entry = @splat(.{}), + mutex: Io.Mutex = .init, + // Shared staging (guarded by `mutex`). + data_buf: [msize]u8 = undefined, + stage_buf: [msize]u8 = undefined, + list_names: [list_name_bytes]u8 = undefined, + list_dirs: [list_capacity]bool = undefined, + list_offs: [list_capacity]u32 = undefined, + + // The derived view (`/active`). An empty `proc` base turns it off, + // which is the default: a test must opt in to reading a process + // tree, and never the live one. + live: active.Table = .{}, + proc_buf: [base_capacity]u8 = @splat(0), + proc_len: u16 = 0, + home_buf: [base_capacity]u8 = @splat(0), + home_len: u16 = 0, + zmx_buf: [base_capacity]u8 = @splat(0), + zmx_len: u16 = 0, + runtime_buf: [base_capacity]u8 = @splat(0), + runtime_len: u16 = 0, + envp: ?cloud9.post.Env = null, + self_pid: u32 = 0, + btime: i64 = 0, + allow_move: bool = false, + act_text: [1024]u8 = undefined, + act_name: [64]u8 = undefined, + act_path: [active.path_capacity]u8 = undefined, + + pub const InitOptions = struct { + io: Io, + pid: u32, + /// Base path of each root, or "" when the root is not pinned. + bases: [5][]const u8, + /// The proc filesystem `/active` reads. Empty — the default — + /// leaves the derived view out of the tree entirely. + proc: []const u8 = "", + /// $HOME, for the harnesses' store-slug spellings. + home: []const u8 = "", + /// The zmx binary a move execs, resolved to an absolute path by + /// the caller. Never client bytes. + zmx: []const u8 = "zmx", + /// $XDG_RUNTIME_DIR, where a move looks for the zmx session it + /// is about to create. + runtime: []const u8 = "", + /// The daemon's own environment block, handed to the harness a + /// move re-execs. Null leaves a moved agent with an empty one. + envp: ?cloud9.post.Env = null, + /// Whether writing `/active///zmx` may move a session. + allow_move: bool = false, + }; + + pub fn init(h: *Harness, o: InitOptions) void { + h.* = .{ .io = o.io, .pid = o.pid, .started_sec = Io.Timestamp.now(o.io, .real).toSeconds() }; + if (o.proc.len > 0 and o.proc.len <= base_capacity) { + @memcpy(h.proc_buf[0..o.proc.len], o.proc); + h.proc_len = @intCast(o.proc.len); + h.self_pid = o.pid; + h.btime = active.bootTime(o.io, o.proc); + } + if (o.home.len <= base_capacity) { + @memcpy(h.home_buf[0..o.home.len], o.home); + h.home_len = @intCast(o.home.len); + } + if (o.zmx.len > 0 and o.zmx.len <= base_capacity) { + @memcpy(h.zmx_buf[0..o.zmx.len], o.zmx); + h.zmx_len = @intCast(o.zmx.len); + } + if (o.runtime.len <= base_capacity) { + @memcpy(h.runtime_buf[0..o.runtime.len], o.runtime); + h.runtime_len = @intCast(o.runtime.len); + } + h.allow_move = o.allow_move; + h.envp = o.envp; + inline for (0..5) |i| { + const path = o.bases[i]; + h.base_set[i] = path.len > 0 and path.len <= base_capacity; + if (h.base_set[i]) { + @memcpy(h.base_buf[i][0..path.len], path); + h.base_len[i] = @intCast(path.len); + } + } + } + + pub fn base(h: *const Harness, r: Root) ?[]const u8 { + if (!h.base_set[@intFromEnum(r)]) return null; + return h.base_buf[@intFromEnum(r)][0..h.base_len[@intFromEnum(r)]]; + } + + // -- path table -- + + fn hashOf(r: Root, rel_path: []const u8) u64 { + var hh = std.hash.Wyhash.init(0x6861_726e_6573_7300); // "harness" + hh.update(&.{@intFromEnum(r)}); + hh.update(rel_path); + return hh.final(); + } + + /// Finds or creates the entry for (root, rel). `ref` says whether the + /// caller hands out a reference (a lookup) or only names the file (a + /// readdir record). Two walks of the same file find the same entry, + /// so their node ids are equal; a freed entry's generation changes. + fn entryFor(h: *Harness, r: Root, rel_path: []const u8, ref: bool) ?*Entry { + if (rel_path.len == 0 or rel_path.len > rel_capacity) return null; + const hash = hashOf(r, rel_path); + var free: ?*Entry = null; + for (&h.table) |*e| { + if (!e.used) { + if (free == null) free = e; + continue; + } + if (e.hash == hash and e.root == r and e.rel_len == rel_path.len and + std.mem.eql(u8, e.rel(), rel_path)) + { + if (ref) e.refs += 1; + return e; + } + } + const e = free orelse return null; + e.used = true; + e.root = r; + e.rel_len = @intCast(rel_path.len); + @memcpy(e.rel_buf[0..rel_path.len], rel_path); + e.hash = hash; + e.refs = @intFromBool(ref); + e.gen +%= 1; + return e; + } + + fn slotOf(h: *Harness, e: *Entry) u12 { + const off = (@intFromPtr(e) - @intFromPtr(&h.table)) / @sizeOf(Entry); + return @intCast(off); + } + + /// The target a node names; a freed or forged serial answers null. + fn resolve(h: *Harness, node: u64) ?Target { + const n: Node = @bitCast(node); + switch (std.enums.fromInt(Kind, n.kind) orelse return null) { + .top => return .{ .top = std.enums.fromInt(Top, n.idx) orelse return null }, + .path => { + const r = std.enums.fromInt(Root, n.idx) orelse return null; + const slot: u12 = @truncate(n.serial); + const e = &h.table[slot]; + if (!e.used or e.root != r) return null; + if (serialOf(slot, e.gen) != n.serial) return null; + return .{ .path = e }; + }, + .active => { + const f = std.enums.fromInt(AFile, n.idx) orelse return null; + if (f == .harness_dir) { + const k = std.enums.fromInt(active.Kind, @as(u8, @truncate(n.serial))) orelse return null; + return .{ .act = .{ .file = f, .kind = k } }; + } + const slot: u8 = @truncate(n.serial); + const gen: u8 = @truncate(n.serial >> 8); + const agent: u8 = @truncate(n.serial >> 16); + const l = h.live.at(slot, gen) orelse return null; + return .{ .act = .{ .file = f, .kind = l.kind, .slot = slot, .gen = gen, .agent = agent } }; + }, + } + } + + fn entryNode(h: *Harness, e: *Entry) u64 { + return nodeOf(e, h.slotOf(e)); + } +}; + +pub const Target = union(enum) { + top: Top, + path: *Entry, + act: Act, +}; + +// ---- attributes --------------------------------------------------------------- + +/// Opens `/` without following a symlink at any step +/// below the base: every component is opened with `O_NOFOLLOW`, so a +/// name swapped for a symlink between the walk and the read fails +/// instead of reaching out of the root. Composing the whole path and +/// opening it in one call cannot do that — only the last component's +/// symlink is refused then, and every directory above it is followed +/// wherever it points. The base itself is configuration (`--root` or +/// $HOME), not client bytes, so it resolves normally. The caller owns +/// the descriptor. +fn openIn(h: *Harness, r: Root, rel_path: []const u8, final: linux.O) ?i32 { + const b = h.base(r) orelse return null; + if (rel_path.len > rel_capacity or b.len > base_capacity) return null; + const walk: linux.O = .{ .PATH = true, .DIRECTORY = true, .CLOEXEC = true, .NOFOLLOW = true }; + + var base_z: [base_capacity + 1]u8 = @splat(0); + @memcpy(base_z[0..b.len], b); + var top_flags = if (rel_path.len == 0) final else walk; + top_flags.NOFOLLOW = false; // the pinned root may legitimately be a link + top_flags.CLOEXEC = true; + var fd = fdOf(linux.open(@ptrCast(&base_z), top_flags, 0)) orelse return null; + if (rel_path.len == 0) return fd; + + var rest = rel_path; + while (true) { + const slash = std.mem.indexOfScalar(u8, rest, '/'); + const comp = if (slash) |i| rest[0..i] else rest; + const last = slash == null; + // Nothing below composes these; a component that could re-enter + // the walk is refused rather than opened. + if (comp.len == 0 or comp.len > 255 or + std.mem.eql(u8, comp, ".") or std.mem.eql(u8, comp, "..")) + { + _ = linux.close(fd); + return null; + } + var name_z: [256]u8 = @splat(0); + @memcpy(name_z[0..comp.len], comp); + var flags = if (last) final else walk; + flags.NOFOLLOW = true; + flags.CLOEXEC = true; + const next = fdOf(linux.openat(fd, @ptrCast(&name_z), flags, 0)); + _ = linux.close(fd); + fd = next orelse return null; + if (last) return fd; + rest = rest[comp.len + 1 ..]; + } +} + +fn fdOf(rc: usize) ?i32 { + if (linux.errno(rc) != .SUCCESS) return null; + return @intCast(rc); +} + +/// Stat of `/`, resolved the no-follow way. A symlink +/// is never served: it is the one escape hatch around the pinned roots. +/// `O_PATH` opens the name itself, so a fifo or device never blocks and +/// never has its open side effects run. +/// A stat plus the file's identity on disk. +const Stat = struct { + st: Io.Dir.Stat, + /// The qid path to report, or 0 when the kernel would not say. + /// + /// intro(5) is strict about what a qid path means: "The path is an + /// integer unique among all files in the hierarchy. If a file is deleted + /// and recreated with the same name in the same directory, the old and + /// new path components of the qids should be different" — and two files + /// are the same file if and only if their qids match. So the path has to + /// be a property of the *file*, stable for as long as it exists. + /// + /// The backend's own node id cannot supply that: it is a slot in the path + /// table, and a slot freed by its last release comes back with a new + /// generation (the guard that stops a forged node id resolving). A client + /// that walks, stats and clunks — `9p stat` does exactly this — saw a + /// different qid path for the same unchanged file every single time, and + /// by the protocol's own rule that means a different file. + /// + /// So this is u9fs's answer instead, from the same fstat the attr already + /// needs: + /// + /// memmove(p, &st->st_ino, ...); *(dev_t*)q ^= st->st_dev; -- u9fs.c + /// + /// The device is mixed in because an inode number is only unique within + /// its filesystem, and the five pinned roots need not share one. The + /// engine keeps resolving requests by node id, which still recycles; only + /// the qid the client sees is stabilised (fs.Attr.path exists for exactly + /// this split, and the engine uses it for nothing else). + qid_path: u64, +}; + +fn statIn(h: *Harness, r: Root, rel_path: []const u8) ?Stat { + const fd = openIn(h, r, rel_path, .{ .PATH = true, .CLOEXEC = true }) orelse return null; + const f: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + defer f.close(h.io); + const st = f.stat(h.io) catch return null; + if (st.kind == .sym_link) return null; // never served: an escape hatch + return .{ .st = st, .qid_path = identityOf(fd) }; +} + +/// (dev, ino) of an open descriptor, folded into one integer. 0 when statx +/// declines to answer, which the caller reads as "fall back to the node id". +fn identityOf(fd: i32) u64 { + var sx: linux.Statx = undefined; + const want: linux.STATX = .{ .INO = true }; + const rc = linux.statx(fd, "", linux.AT.EMPTY_PATH, want, &sx); + if (linux.errno(rc) != .SUCCESS or !sx.mask.INO) return 0; + const dev = (@as(u64, sx.dev_major) << 32) | sx.dev_minor; + // The inode carries the entropy, so it stays unshifted and the device is + // folded over its top bits: two roots on one filesystem keep distinct + // inodes, and the same inode on two filesystems stops colliding. + return sx.ino ^ (dev *% 0x9E37_79B9_7F4A_7C15); +} + +/// The qid version of a file on disk, the way u9fs derives it: +/// +/// qid.vers = st->st_mtime ^ (st->st_size << 8); -- u9fs.c +/// +/// intro(5) makes the qid the server's identity for a file — "two files on +/// the same server hierarchy are the same if and only if their qids are the +/// same" — and the version the part that "is incremented every time the file +/// is modified". A server that leaves it 0 is telling every client that +/// nothing it serves ever changes, which is exactly wrong for this tree: its +/// whole point is transcripts that grow while you read them. The size shift +/// is what makes an append within the same second still change the version, +/// and a rewrite that keeps the size change it through the mtime. +fn qidVers(mtime_sec: i64, size: u64) u32 { + return @truncate(@as(u64, @bitCast(mtime_sec)) ^ (size << 8)); +} + +/// The qid version of text this server derives rather than reads: there is no +/// mtime behind it, so the content itself is the only honest version. +fn textVers(text: []const u8) u32 { + return @truncate(std.hash.Wyhash.hash(0, text)); +} + +fn statAttr(h: *Harness, e: *Entry) ?fs.Attr { + const got = statIn(h, e.root, e.rel()) orelse return null; + const st = got.st; + const mtime = st.mtime.toSeconds(); + const size: u64 = if (st.kind == .directory) 0 else st.size; + return .{ + .name = basename(e.rel()), + .node = h.entryNode(e), + .dir = st.kind == .directory, + // stat(5): "Directories and most files representing devices have a + // conventional length of 0." + .size = size, + .mode = if (st.kind == .directory) 0o555 else 0o444, + .mtime = @truncate(@as(u64, @bitCast(mtime))), + .version = qidVers(mtime, size), + .path = if (got.qid_path != 0) got.qid_path else null, + }; +} + +fn basename(rel_path: []const u8) []const u8 { + if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| return rel_path[i + 1 ..]; + return rel_path; +} + +/// Served as `/README`: what the tree is and how to read it. Static, so it +/// is the one "fact" that needs no buffer — `factText` returns it directly. +const readme_text = + \\9agents - every coding agent's state on this machine, as files. + \\Read-only: writes answer EPERM. Nothing is cached, so a transcript + \\grows while you cat it and a new session appears as soon as its + \\file lands. + \\ + \\ README this file + \\ pid uptime this daemon's pid; seconds since it started + \\ active/ what is running right now + \\ claude/ projects/ history skills/ + \\ codex/ sessions/ session-index history + \\ omp/ hermes/ dsh/ each harness's own directory, raw + \\ skills/ claude/ codex/ omp/, in one place + \\ + \\active/// holds pid ppid started cwd name title + \\session via status transcript zmx. + \\ session its session id, empty when it did not resolve + \\ via how that id was found: registry, fd, dir or none + \\ status only harnesses that publish one have it + \\ transcript the live .jsonl; its mtime is the last activity + \\ zmx the zmx session it runs in, empty outside zmx + \\ + \\ ls active/*/* what is running + \\ cat active/claude/345104/session what it is + \\ tail -f active/claude/345104/transcript + \\ + \\Credentials are excluded by name and never listed, at any depth; + \\symlinks are never served. Writing a name into an agent's zmx file + \\moves it into that zmx session, and only when the daemon was + \\started with --allow-move; otherwise that write answers EPERM too. + \\ +; + +/// A fact's text, rendered on demand. `buf` should be a small stack buffer. +fn factText(h: *Harness, fact: Top, buf: []u8) []const u8 { + if (fact == .readme) return readme_text; // static: outlives any buffer + var w = Io.Writer.fixed(buf); + switch (fact) { + .pid => w.print("{d}\n", .{h.pid}) catch {}, + .uptime => { + const now = Io.Timestamp.now(h.io, .real).toSeconds(); + const up: u64 = if (now > h.started_sec) @intCast(now - h.started_sec) else 0; + w.print("{d}\n", .{up}) catch {}; + }, + else => {}, + } + return w.buffered(); +} + +var fact_buf: [64]u8 = undefined; + +fn attrFor(h: *Harness, t: Target) ?fs.Attr { + switch (t) { + .top => |top| { + switch (top) { + .root => return .{ .name = "/", .node = root, .dir = true, .mode = 0o555 }, + .active => return .{ .name = "active", .node = topNode(.active), .dir = true, .mode = 0o555 }, + .pid, .uptime, .readme => return .{ + .name = top.fileName(), + .node = topNode(top), + .size = factText(h, top, &fact_buf).len, + .mode = 0o444, + }, + .claude, .codex, .skills => return .{ + .name = top.fileName(), + .node = topNode(top), + .dir = true, + .mode = 0o555, + }, + .omp, .hermes, .dsh => { + // The harness dir IS its mirror: stat the target. + const m = top.mirror().?; + const got = statIn(h, m.root, m.rel) orelse return null; + if (got.st.kind != .directory) return null; + return .{ + .name = top.fileName(), + .node = topNode(top), + .dir = true, + .mode = 0o555, + .path = if (got.qid_path != 0) got.qid_path else null, + }; + }, + } + }, + .path => |e| return statAttr(h, e), + .act => |a| return activeAttr(h, a), + } +} + +/// Attr of the mount target as a table entry (fresh from disk, or null +/// when the target is missing, a symlink, or excluded by name). +fn attrOfMount(h: *Harness, r: Root, rel_path: []const u8) ?fs.Attr { + const e = h.entryFor(r, rel_path, false) orelse return null; + defer if (e.refs == 0) { + e.used = false; + e.gen +%= 1; + }; + return statAttr(h, e); +} + +// ---- /active: the derived view ------------------------------------------------- + +fn sourcesOf(h: *Harness) active.Sources { + var roots: [5][]const u8 = @splat(""); + for (0..5) |i| { + if (h.base_set[i]) roots[i] = h.base_buf[i][0..h.base_len[i]]; + } + return .{ + .io = h.io, + .proc = h.proc_buf[0..h.proc_len], + .home = h.home_buf[0..h.home_len], + .roots = roots, + .self_pid = h.self_pid, + .btime = h.btime, + }; +} + +/// Rescans the process tree. Done on a listing of `/active` and on a +/// lookup that misses, which is what makes an agent visible the moment +/// it starts and gone the moment it exits. +fn rescan(h: *Harness) void { + if (h.proc_len == 0) return; + h.live.scan(sourcesOf(h)); +} + +fn hasLive(h: *Harness, k: active.Kind) bool { + for (&h.live.slots) |*sl| { + if (sl.used and sl.live.kind == k) return true; + } + return false; +} + +/// `\n` staged where an answer can point at it. +fn line(h: *Harness, text: []const u8) ?[]const u8 { + return std.fmt.bufPrint(&h.act_text, "{s}\n", .{text}) catch null; +} + +/// The value of a field file, or null when this agent has none — an +/// absent field is not listed and does not resolve, so the tree never +/// answers a blank where it does not know. +fn activeText(h: *Harness, a: Act) ?[]const u8 { + const l = h.live.at(a.slot, a.gen) orelse return null; + const src = sourcesOf(h); + var scratch: [active.text_capacity]u8 = undefined; + return switch (a.file) { + .pid => std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.pid}) catch null, + .ppid => if (l.ppid == 0) null else std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.ppid}) catch null, + .started => if (l.started == 0) null else std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.started}) catch null, + .cwd => if (l.cwd.len == 0) null else line(h, l.cwd.slice()), + .session => if (l.session.len == 0) null else line(h, l.session.slice()), + .via => line(h, l.via.text()), + .name => line(h, active.registryField(src, l, "name", &scratch) orelse return null), + .status => line(h, active.registryField(src, l, "status", &scratch) orelse return null), + .title => line(h, active.headField(h.io, l.kind, l.transcript.slice(), .title, &scratch) orelse return null), + .model => line(h, active.headField(h.io, l.kind, l.transcript.slice(), .model, &scratch) orelse return null), + .agent_model => blk: { + const path = activePath(h, a) orelse break :blk null; + break :blk line(h, active.headField(h.io, l.kind, path, .model, &scratch) orelse return null); + }, + // `zmx` is always there, so it can always be written to; empty + // means the agent runs outside zmx. + .zmx => if (active.zmxOf(src, l.pid, &scratch)) |v| line(h, v) else h.act_text[0..0], + else => null, + }; +} + +/// The absolute path behind a transcript-shaped file. +fn activePath(h: *Harness, a: Act) ?[]const u8 { + const l = h.live.at(a.slot, a.gen) orelse return null; + switch (a.file) { + .transcript => return if (l.transcript.len == 0) null else l.transcript.slice(), + .agent_transcript, .agent_model => { + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + return active.agentPath(l, &ag, a.agent, &h.act_path); + }, + else => return null, + } +} + +/// Opens an absolute path without following a symlink at the last +/// component. These paths are derived from a pinned root or from the +/// fd the harness itself holds, never from client bytes, but a name +/// swapped underneath must still fail rather than redirect. +fn openAbsNoFollow(path: []const u8) ?i32 { + if (path.len == 0 or path.len >= active.path_capacity) return null; + var z: [active.path_capacity]u8 = @splat(0); + @memcpy(z[0..path.len], path); + const rc = linux.open(@ptrCast(&z), .{ + .ACCMODE = .RDONLY, + .NOFOLLOW = true, + .CLOEXEC = true, + .NONBLOCK = true, + }, 0); + if (linux.errno(rc) != .SUCCESS) return null; + return @intCast(rc); +} + +fn activeAttr(h: *Harness, a: Act) ?fs.Attr { + switch (a.file) { + .harness_dir => { + if (!hasLive(h, a.kind)) return null; + return .{ .name = a.kind.text(), .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .entry => { + const l = h.live.at(a.slot, a.gen) orelse return null; + const nm = std.fmt.bufPrint(&h.act_name, "{d}", .{l.pid}) catch return null; + return .{ .name = nm, .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .agents => { + const l = h.live.at(a.slot, a.gen) orelse return null; + if (l.agent_dir.len == 0) return null; + return .{ .name = "agents", .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .agent => { + const l = h.live.at(a.slot, a.gen) orelse return null; + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + if (a.agent >= ag.count) return null; + const nm = ag.name(a.agent); + @memcpy(h.act_name[0..nm.len], nm); + return .{ .name = h.act_name[0..nm.len], .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .transcript, .agent_transcript => { + const path = activePath(h, a) orelse return null; + const st = Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = false }) catch return null; + if (st.kind != .file) return null; + const mtime = st.mtime.toSeconds(); + return .{ + .name = a.file.fileName(), + .node = activeNode(a), + .size = st.size, + .mode = 0o444, + .mtime = @truncate(@as(u64, @bitCast(mtime))), + .version = qidVers(mtime, st.size), + }; + }, + else => { + const text = activeText(h, a) orelse return null; + // Only `zmx` is ever writable, and only when a move is allowed. + const mode: u16 = if (a.file == .zmx and h.allow_move) 0o644 else 0o444; + return .{ + .name = a.file.fileName(), + .node = activeNode(a), + .size = text.len, + .mode = mode, + // Derived, not read: `status` flipping busy->idle keeps its + // size, so only the text can carry the change. + .version = textVers(text), + }; + }, + } +} + +/// The slot holding `pid` for this harness, if any. +fn slotOfPid(h: *Harness, k: active.Kind, pid: u32) ?Act { + for (&h.live.slots, 0..) |*sl, i| { + if (!sl.used or sl.live.kind != k or sl.live.pid != pid) continue; + return .{ .file = .entry, .kind = k, .slot = @intCast(i), .gen = sl.gen }; + } + return null; +} + +fn activeLookup(h: *Harness, req: fs.Req, a: Act, name: []const u8) Answer { + switch (a.file) { + .harness_dir => { + const pid = std.fmt.parseInt(u32, name, 10) catch return fail(req.tag, E.NOENT); + var found = slotOfPid(h, a.kind, pid); + if (found == null) { + rescan(h); + found = slotOfPid(h, a.kind, pid); + } + const act = found orelse return fail(req.tag, E.NOENT); + return activeReply(h, req, act, name); + }, + .entry => { + if (std.mem.eql(u8, name, "agents")) { + return activeReply(h, req, .{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen }, name); + } + for (AFile.entry_files) |f| { + if (!std.mem.eql(u8, f.fileName(), name)) continue; + return activeReply(h, req, .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen }, name); + } + return fail(req.tag, E.NOENT); + }, + .agents => { + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + const i = ag.indexOf(name) orelse return fail(req.tag, E.NOENT); + return activeReply(h, req, .{ + .file = .agent, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = @intCast(i), + }, name); + }, + .agent => { + for (AFile.agent_files) |f| { + if (!std.mem.eql(u8, f.fileName(), name)) continue; + return activeReply(h, req, .{ + .file = f, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = a.agent, + }, name); + } + return fail(req.tag, E.NOENT); + }, + else => return fail(req.tag, E.NOTDIR), + } +} + +fn activeReply(h: *Harness, req: fs.Req, a: Act, name: []const u8) Answer { + const attr = activeAttr(h, a) orelse return fail(req.tag, E.NOENT); + var with_name = attr; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; +} + +fn activeReaddir(h: *Harness, req: fs.Req, a: Act) Answer { + var st: Staging = .{ .buf = &h.stage_buf, .skip = req.off }; + switch (a.file) { + .harness_dir => { + for (&h.live.slots, 0..) |*sl, i| { + if (!sl.used or sl.live.kind != a.kind) continue; + var num: [24]u8 = undefined; + const nm = std.fmt.bufPrint(&num, "{d}", .{sl.live.pid}) catch continue; + st.add(activeNode(.{ + .file = .entry, + .kind = a.kind, + .slot = @intCast(i), + .gen = sl.gen, + }), true, nm); + } + }, + .entry => { + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + for (AFile.entry_files) |f| { + const child: Act = .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen }; + if (activeAttr(h, child) == null) continue; // no value: not listed + st.add(activeNode(child), false, f.fileName()); + } + if (l.agent_dir.len > 0) { + st.add(activeNode(.{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen }), true, "agents"); + } + }, + .agents => { + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + for (0..ag.count) |i| { + st.add(activeNode(.{ + .file = .agent, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = @intCast(i), + }), true, ag.name(i)); + } + }, + .agent => { + for (AFile.agent_files) |f| { + const child: Act = .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen, .agent = a.agent }; + if (activeAttr(h, child) == null) continue; + st.add(activeNode(child), false, f.fileName()); + } + }, + else => return fail(req.tag, E.NOTDIR), + } + return .{ .reply = .{ .tag = req.tag }, .bytes = h.stage_buf[0..st.len] }; +} + +fn activeRead(h: *Harness, req: fs.Req, a: Act) Answer { + switch (a.file) { + .harness_dir, .entry, .agents, .agent => return fail(req.tag, E.ISDIR), + .transcript, .agent_transcript => { + const path = activePath(h, a) orelse return fail(req.tag, E.NOENT); + const fd = openAbsNoFollow(path) orelse return fail(req.tag, E.NOENT); + const file: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + defer file.close(h.io); + const st = file.stat(h.io) catch return fail(req.tag, E.IO); + if (st.kind == .directory) return fail(req.tag, E.ISDIR); + if (st.kind != .file) return fail(req.tag, E.PERM); + _ = linux.fcntl(fd, linux.F.SETFL, 0); + const want = @min(req.size, h.data_buf.len); + const n = file.readPositionalAll(h.io, h.data_buf[0..want], req.off) catch + return fail(req.tag, E.IO); + return .{ .reply = .{ .tag = req.tag }, .bytes = h.data_buf[0..n] }; + }, + else => { + const text = activeText(h, a) orelse return fail(req.tag, E.NOENT); + return window(req, text); + }, + } +} + +// ---- the move: writing a zmx session name into an agent's `zmx` ---------------- + +/// How long a move waits, in 100ms ticks: for the agent to take the +/// hangup, then to die outright, then for the new zmx session to post. +const term_ticks: usize = 50; +const kill_ticks: usize = 30; +const post_ticks: usize = 100; + +fn napOneTick(h: *Harness) void { + h.io.sleep(.fromMilliseconds(100), .awake) catch {}; +} + +/// Is this name already a live zmx session? zmx posts each session into +/// the registry, so the registry is the answer — no process scanning. +fn zmxPosted(h: *Harness, name: []const u8) bool { + if (h.runtime_len == 0) return false; + var buf: [active.path_capacity]u8 = undefined; + const path = std.fmt.bufPrint(&buf, "{s}/9p/zmx/{s}", .{ h.runtime_buf[0..h.runtime_len], name }) catch return true; + return Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = true }) != error.FileNotFound; +} + +fn procGone(h: *Harness, pid: u32) bool { + var buf: [active.path_capacity]u8 = undefined; + const path = std.fmt.bufPrint(&buf, "{s}/{d}", .{ h.proc_buf[0..h.proc_len], pid }) catch return false; + _ = Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = true }) catch return true; + return false; +} + +/// SIGTERM, then SIGKILL, then give up. The agent's session was +/// resolved before this ran, so whatever happens it has somewhere to +/// come back to. +fn killAndWait(h: *Harness, pid: u32) bool { + _ = linux.kill(@intCast(pid), .TERM); + for (0..term_ticks) |_| { + if (procGone(h, pid)) return true; + napOneTick(h); + } + _ = linux.kill(@intCast(pid), .KILL); + for (0..kill_ticks) |_| { + if (procGone(h, pid)) return true; + napOneTick(h); + } + return false; +} + +/// `zmx run -d `, in the +/// agent's own directory. Every word but `` is fixed by the +/// harness; `` was checked against zmx's label charset before +/// anything was killed. No byte a client wrote reaches `exec` as a +/// command. +fn spawnZmx(h: *Harness, l: *const active.Live, name: []const u8, resume_value: []const u8) bool { + if (h.zmx_len == 0) return false; + const harness_argv0: []const u8 = @tagName(l.kind); + const resume_flag: []const u8 = switch (l.kind) { + .codex => "resume", + else => "--resume", + }; + + // One buffer holds every NUL-terminated word; `argv` points into it. + var words: [8 * active.path_capacity]u8 = undefined; + var used: usize = 0; + var argv: [9]?[*:0]const u8 = @splat(null); + var n: usize = 0; + const parts = [_][]const u8{ + h.zmx_buf[0..h.zmx_len], "run", name, "-d", + harness_argv0, resume_flag, resume_value, + }; + for (parts) |w| { + if (used + w.len + 1 > words.len or n + 1 >= argv.len) return false; + @memcpy(words[used..][0..w.len], w); + words[used + w.len] = 0; + argv[n] = @ptrCast(&words[used]); + n += 1; + used += w.len + 1; + } + + var cwd_z: [active.cwd_capacity + 1]u8 = @splat(0); + if (l.cwd.len >= cwd_z.len) return false; + @memcpy(cwd_z[0..l.cwd.len], l.cwd.slice()); + + const rc = linux.fork(); + if (linux.errno(rc) != .SUCCESS) return false; + if (rc == 0) { + // The child: only async-signal-safe calls from here. + _ = linux.chdir(@ptrCast(&cwd_z)); + const empty = [_:null]?[*:0]const u8{}; + const envp: cloud9.post.Env = h.envp orelse ∅ + _ = linux.execve(argv[0].?, @ptrCast(&argv), envp); + linux.exit(127); + } + // Reap the forked `zmx`, which returns as soon as the session is up. + const child: i32 = @intCast(rc); + var status: u32 = 0; + for (0..post_ticks) |_| { + const w = linux.waitpid(child, &status, 1); // WNOHANG + if (w == @as(usize, @intCast(child))) break; + napOneTick(h); + } + return true; +} + +/// A move: kill the agent and bring it back inside a zmx session of the +/// name written. Refusals come before anything is destroyed. +fn moveToZmx(h: *Harness, req: fs.Req, a: Act) Answer { + if (!h.allow_move) return failWhy(req.tag, E.PERM, "moves are not permitted: start 9agents with --allow-move"); + const name = std.mem.trim(u8, req.data, " \t\r\n"); + if (!active.legalZmxName(name)) return failWhy(req.tag, E.INVAL, "invalid zmx name: letters, digits, dot, dash and underscore only"); + + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + // Nothing is killed that has nowhere to come back to. + if (l.via == .none or l.session.len == 0) return failWhy(req.tag, E.PERM, "no session resolved: not permitted, there would be nothing to resume"); + if (l.cwd.len == 0) return failWhy(req.tag, E.PERM, "no working directory known for this agent: not permitted"); + const resume_value: []const u8 = switch (l.kind) { + // omp resumes by the transcript it wrote, the others by id. + .omp => l.transcript.slice(), + .claude, .codex, .hermes => l.session.slice(), + // dsh has no resume form worth guessing at. + .dsh => return failWhy(req.tag, E.PERM, "dsh has no resume form: moving it is not permitted"), + }; + if (resume_value.len == 0) return failWhy(req.tag, E.PERM, "nothing to resume from: not permitted"); + + // Already there: setting a value it already has changes nothing. + var have: [active.text_capacity]u8 = undefined; + if (active.zmxOf(sourcesOf(h), l.pid, &have)) |current| { + if (std.mem.eql(u8, current, name)) { + return .{ .reply = .{ .tag = req.tag, .written = @intCast(req.data.len) } }; + } + } + if (zmxPosted(h, name)) return failWhy(req.tag, E.EXIST, "a live zmx session of that name already exists"); + // The slot could have gone stale between the scan and this write. + if (!active.stillAlive(sourcesOf(h), l)) return failWhy(req.tag, E.NOENT, "the agent exited before the move began: no such process"); + + // Everything below this line destroys something. + var snapshot = l.*; + if (!killAndWait(h, snapshot.pid)) return failWhy(req.tag, E.IO, "the agent did not exit when signalled; nothing was started"); + if (!spawnZmx(h, &snapshot, name, resume_value)) return failWhy(req.tag, E.IO, "could not start zmx; the agent has already been stopped"); + for (0..post_ticks) |_| { + if (zmxPosted(h, name)) { + rescan(h); + return .{ .reply = .{ .tag = req.tag, .written = @intCast(req.data.len) } }; + } + napOneTick(h); + } + rescan(h); + return failWhy(req.tag, E.IO, "zmx did not post the session in time; the agent may still return"); +} + +// ---- dispatch ------------------------------------------------------------------ + +/// Answers one engine request. The caller holds `h.mutex` and replies +/// before letting go of it (answer bytes point into `h`). +pub fn handle(h: *Harness, req: fs.Req) Answer { + const t = h.resolve(req.node) orelse return fail(req.tag, E.NOENT); + return switch (req.op) { + .lookup => lookup(h, req, t), + .getattr => attrReply(h, req.tag, t), + .setattr => setattrReq(h, req, t), + .open => open(h, req, t), + .release => release(h, req, t), + .readdir => readdir(h, req, t), + .read => read(h, req, t), + .write => writeReq(h, req, t), + }; +} + +/// Truncation is how a shell's `>` opens a file before writing it. The +/// `zmx` file has no length of its own — a write replaces the value — +/// so on the one writable file a zero-length truncate is a no-op +/// rather than a refusal. Everything else still answers EPERM. +fn setattrReq(h: *Harness, req: fs.Req, t: Target) Answer { + switch (t) { + .act => |a| if (a.file == .zmx and h.allow_move and req.truncate) { + return .{ .reply = .{ .tag = req.tag } }; + }, + else => {}, + } + return failWhy(req.tag, E.PERM, "9agents serves a view, not a store: writing is not permitted"); +} + +/// The tree answers EPERM to every write but one: a zmx session name +/// into a live agent's `zmx`, which moves it there. +fn writeReq(h: *Harness, req: fs.Req, t: Target) Answer { + switch (t) { + .act => |a| if (a.file == .zmx) return moveToZmx(h, req, a), + else => {}, + } + return failWhy(req.tag, E.PERM, "9agents serves a view, not a store: writing is not permitted"); +} + +fn attrReply(h: *Harness, tag: u64, t: Target) Answer { + const a = attrFor(h, t) orelse return fail(tag, E.NOENT); + return .{ .reply = .{ .tag = tag, .attr = a } }; +} + +/// A lookup's attr names the file as the client asked for it (the engine +/// keeps that name for the fid); refs for path targets were already taken +/// by `entryFor`. +fn lookupAttr(h: *Harness, req: fs.Req, t: Target, name: []const u8) Answer { + const a = attrFor(h, t) orelse return fail(req.tag, E.NOENT); + var with_name = a; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; +} + +fn lookup(h: *Harness, req: fs.Req, t: Target) Answer { + const name = req.data; + if (name.len == 0 or name.len > 255) return fail(req.tag, E.NOENT); + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return fail(req.tag, E.NOENT); + + if (std.mem.eql(u8, name, ".")) { + // The engine asks for "." when cloning a fid; the reference is + // paid for path targets here like any other lookup result. + switch (t) { + .top, .act => return attrReply(h, req.tag, t), + .path => |e| { + e.refs += 1; + return lookupAttr(h, req, t, name); + }, + } + } + if (std.mem.eql(u8, name, "..")) return lookupParent(h, req, t); + + switch (t) { + .top => |top| switch (top) { + .root => { + const found: ?Top = blk: for (Top.listed) |e| { + if (std.mem.eql(u8, e.fileName(), name)) break :blk e; + } else break :blk null; + return attrReply(h, req.tag, .{ .top = found orelse return fail(req.tag, E.NOENT) }); + }, + .claude, .codex, .skills => { + const m = mountNamed(top, name) orelse return fail(req.tag, E.NOENT); + const e = h.entryFor(m.root, m.rel, true) orelse return fail(req.tag, E.NFILE); + const a = statAttr(h, e) orelse { + unref(e); + return fail(req.tag, E.NOENT); + }; + var with_name = a; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; + }, + .omp, .hermes, .dsh => { + const m = top.mirror().?; + return lookupBelow(h, req, m.root, m.rel, name); + }, + .active => { + const k = blk: for (std.enums.values(active.Kind)) |k| { + if (std.mem.eql(u8, k.text(), name)) break :blk k; + } else return fail(req.tag, E.NOENT); + if (!hasLive(h, k)) { + rescan(h); + if (!hasLive(h, k)) return fail(req.tag, E.NOENT); + } + return activeReply(h, req, .{ .file = .harness_dir, .kind = k }, name); + }, + .pid, .uptime, .readme => return fail(req.tag, E.NOTDIR), + }, + .path => |e| { + const rel_path = e.rel(); + if (attrFor(h, t)) |a| { + if (!a.dir) return fail(req.tag, E.NOTDIR); + } else return fail(req.tag, E.NOENT); + return lookupBelow(h, req, e.root, rel_path, name); + }, + .act => |a| return activeLookup(h, req, a, name), + } +} + +/// A lookup of `name` in the directory (root, dir_rel): the child's rel +/// path is composed only from the remembered pair and the (single-segment, +/// engine-vetted) name. +fn lookupBelow(h: *Harness, req: fs.Req, r: Root, dir_rel: []const u8, name: []const u8) Answer { + if (excluded(name)) return fail(req.tag, E.NOENT); + var scratch: [rel_capacity + 256]u8 = undefined; + const child_rel = joinRel(&scratch, dir_rel, name) orelse return fail(req.tag, E.NOENT); + const e = h.entryFor(r, child_rel, true) orelse return fail(req.tag, E.NFILE); + const a = statAttr(h, e) orelse { + unref(e); + return fail(req.tag, E.NOENT); + }; + var with_name = a; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; +} + +fn joinRel(scratch: []u8, dir_rel: []const u8, name: []const u8) ?[]const u8 { + if (dir_rel.len == 0) { + if (name.len > scratch.len) return null; + @memcpy(scratch[0..name.len], name); + return scratch[0..name.len]; + } + const total = dir_rel.len + 1 + name.len; + if (total > scratch.len) return null; + @memcpy(scratch[0..dir_rel.len], dir_rel); + scratch[dir_rel.len] = '/'; + @memcpy(scratch[dir_rel.len + 1 .. total], name); + return scratch[0..total]; +} + +fn lookupParent(h: *Harness, req: fs.Req, t: Target) Answer { + const parent: Target = switch (t) { + .top => |top| .{ .top = top.parent() }, + .path => |e| blk: { + const rel_path = e.rel(); + if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| { + const up = rel_path[0..i]; + const pe = h.entryFor(e.root, up, true) orelse return fail(req.tag, E.NFILE); + break :blk .{ .path = pe }; + } + break :blk .{ .top = mountOwner(e.root, rel_path) }; + }, + .act => |a| switch (a.file) { + .harness_dir => .{ .top = .active }, + .entry => .{ .act = .{ .file = .harness_dir, .kind = a.kind } }, + .agents => .{ .act = .{ .file = .entry, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, + .agent => .{ .act = .{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, + .agent_model, .agent_transcript => .{ .act = .{ + .file = .agent, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = a.agent, + } }, + // Every other file hangs directly off its entry. + else => .{ .act = .{ .file = .entry, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, + }, + }; + return attrReply(h, req.tag, parent); +} + +fn unref(e: *Entry) void { + if (e.refs > 0) e.refs -= 1; + if (e.refs == 0) { + e.used = false; + e.gen +%= 1; + } +} + +fn open(h: *Harness, req: fs.Req, t: Target) Answer { + _ = attrFor(h, t) orelse return fail(req.tag, E.NOENT); // still there? + // Read-only tree, with exactly one exception: the `zmx` file of a + // live agent, and only when the daemon was started to allow moves. + // Truncation is meaningless there (a write replaces the value), so + // it is accepted rather than refused, which is what `>` needs. + const writable = switch (t) { + .act => |a| a.file == .zmx and h.allow_move, + else => false, + }; + const rw = req.omode & 3; + if (!writable) { + if (rw == cloud9.owrite or rw == cloud9.ordwr) return failWhy(req.tag, E.PERM, "9agents serves a view, not a store: writing is not permitted"); + if (req.omode & cloud9.otrunc != 0) return failWhy(req.tag, E.PERM, "truncation is not permitted here"); + } + return .{ .reply = .{ .tag = req.tag, .handle = 1 } }; +} + +fn release(h: *Harness, req: fs.Req, t: Target) Answer { + _ = h; + switch (t) { + .top, .act => {}, + .path => |e| unref(e), + } + return .{ .reply = .{ .tag = req.tag } }; +} + +// ---- reads ---------------------------------------------------------------------- + +fn read(h: *Harness, req: fs.Req, t: Target) Answer { + switch (t) { + .top => |top| switch (top) { + .pid, .uptime, .readme => { + const text = factText(h, top, &fact_buf); + return window(req, text); + }, + else => return fail(req.tag, E.ISDIR), + }, + .act => |a| return activeRead(h, req, a), + .path => |e| { + // `O_NONBLOCK` so a fifo left in a harness root cannot park the + // daemon in `open`; the kind check below refuses it anyway. + const fd = openIn(h, e.root, e.rel(), .{ + .ACCMODE = .RDONLY, + .NONBLOCK = true, + .CLOEXEC = true, + }) orelse return fail(req.tag, E.NOENT); + const file: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + defer file.close(h.io); + const st = file.stat(h.io) catch return fail(req.tag, E.IO); + if (st.kind == .directory) return fail(req.tag, E.ISDIR); + // Only regular files have bytes this tree promises to serve. + if (st.kind != .file) return fail(req.tag, E.PERM); + _ = linux.fcntl(fd, linux.F.SETFL, 0); // pread wants no O_NONBLOCK + const want = @min(req.size, h.data_buf.len); + const n = file.readPositionalAll(h.io, h.data_buf[0..want], req.off) catch + return fail(req.tag, E.IO); + return .{ .reply = .{ .tag = req.tag }, .bytes = h.data_buf[0..n] }; + }, + } +} + +/// `text` windowed by the request's offset and size. +fn window(req: fs.Req, text: []const u8) Answer { + const off: usize = @intCast(@min(req.off, text.len)); + const n = @min(text.len - off, req.size); + return .{ .reply = .{ .tag = req.tag }, .bytes = text[off..][0..n] }; +} + +// ---- readdir -------------------------------------------------------------------- + +/// Directory records in the engine's shape: `node:u64le dir:u8 len:u8 name`. +const Staging = struct { + buf: []u8, + len: usize = 0, + skip: u64, + /// Set once a record did not fit. Everything after it is left for the + /// next read: dropping one record and staging a shorter one behind it + /// would lose that entry, because the client's next offset counts the + /// records it received. + full: bool = false, + + fn add(s: *Staging, node: u64, dir: bool, name: []const u8) void { + if (s.full) return; + if (s.skip > 0) { + s.skip -= 1; + return; + } + if (name.len == 0 or name.len > 255) return; + if (s.len + 10 + name.len > s.buf.len) { + s.full = true; + return; + } + std.mem.writeInt(u64, s.buf[s.len..][0..8], node, .little); + s.buf[s.len + 8] = @intFromBool(dir); + s.buf[s.len + 9] = @intCast(name.len); + @memcpy(s.buf[s.len + 10 ..][0..name.len], name); + s.len += 10 + name.len; + } +}; + +fn readdir(h: *Harness, req: fs.Req, t: Target) Answer { + var st: Staging = .{ .buf = &h.stage_buf, .skip = req.off }; + switch (t) { + .top => |top| switch (top) { + .root => for (Top.listed) |e| st.add(topNode(e), e.dir(), e.fileName()), + .claude, .codex, .skills => { + for (named_mounts) |m| { + if (m.owner != top) continue; + // Only mounts whose target exists are listed (a harness + // without skills simply has no skills/ entry). + const a = attrOfMount(h, m.mount.root, m.mount.rel) orelse continue; + const e = h.entryFor(m.mount.root, m.mount.rel, false) orelse continue; + st.add(h.entryNode(e), a.dir, m.name); + } + }, + .pid, .uptime, .readme => return fail(req.tag, E.NOTDIR), + .active => { + rescan(h); + for (std.enums.values(active.Kind)) |k| { + if (!hasLive(h, k)) continue; + st.add(activeNode(.{ .file = .harness_dir, .kind = k }), true, k.text()); + } + }, + .omp, .hermes, .dsh => { + const m = top.mirror().?; + if (h.base(m.root) == null) return fail(req.tag, E.NOENT); + switch (listDir(h, &st, m.root, m.rel)) { + .ok => {}, + .gone => return fail(req.tag, E.NOENT), + .failed => return fail(req.tag, E.IO), + .overflow => return failWhy(req.tag, E.NFILE, "directory has more entries than this server can list at once"), + } + }, + }, + .path => |e| { + if (attrFor(h, t)) |a| { + if (!a.dir) return fail(req.tag, E.NOTDIR); + } else return fail(req.tag, E.NOENT); + switch (listDir(h, &st, e.root, e.rel())) { + .ok => {}, + .gone => return fail(req.tag, E.NOENT), + .failed => return fail(req.tag, E.IO), + .overflow => return failWhy(req.tag, E.NFILE, "directory has more entries than this server can list at once"), + } + }, + .act => |a| return activeReaddir(h, req, a), + } + return .{ .reply = .{ .tag = req.tag }, .bytes = h.stage_buf[0..st.len] }; +} + +/// How a listing attempt ended. `overflow` is deliberate: a comptime cap +/// that would silently drop entries answers an error instead, because a +/// short listing is indistinguishable from a small directory. +const Listing = enum { ok, gone, failed, overflow }; + +/// Stages the contents of the directory (root, dir_rel): every name the +/// walker yields that survives the exclusions, sorted so a listing that +/// spans several reads stays consistent. +fn listDir(h: *Harness, st: *Staging, r: Root, dir_rel: []const u8) Listing { + const fd = openIn(h, r, dir_rel, .{ + .ACCMODE = .RDONLY, + .DIRECTORY = true, + .CLOEXEC = true, + }) orelse return .gone; + const dir: Io.Dir = .{ .handle = fd }; + defer Io.Dir.close(dir, h.io); + var read_buf: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var reader = Io.Dir.Reader.init(dir, &read_buf); + var names_len: usize = 0; + var count: usize = 0; + while (true) { + const entry = (reader.next(h.io) catch return .failed) orelse break; + if (excluded(entry.name)) continue; + var kind = entry.kind; + if (kind == .unknown or kind == .sym_link) { + // The walker's word is not proof: stat without following. + var child_buf: [rel_capacity + 256]u8 = undefined; + const child_rel = joinRel(&child_buf, dir_rel, entry.name) orelse continue; + const cst = statIn(h, r, child_rel) orelse continue; + kind = cst.st.kind; + } + if (kind == .sym_link) continue; // never served + // A cap reached is an error, never a short listing: a directory + // that quietly loses entries is a wrong answer, and a caller + // cannot tell it from a small directory. + if (count == list_capacity or names_len + entry.name.len > h.list_names.len) return .overflow; + @memcpy(h.list_names[names_len..][0..entry.name.len], entry.name); + h.list_offs[count] = @intCast(names_len); + h.list_dirs[count] = kind == .directory; + names_len += entry.name.len; + count += 1; + } + // Sort the (offset, length) pairs by name for cursor stability. + const SortCtx = struct { + names: []const u8, + offs: []const u32, + lens: [list_capacity]u32, + + fn lessThan(ctx: @This(), a: usize, b: usize) bool { + return std.mem.order(u8, ctx.nameAt(a), ctx.nameAt(b)) == .lt; + } + fn nameAt(ctx: @This(), i: usize) []const u8 { + const start = ctx.offs[i]; + return ctx.names[start..][0..ctx.lens[i]]; + } + }; + var lens: [list_capacity]u32 = @splat(0); + var order: [list_capacity]usize = @splat(0); + { + var end: usize = 0; + for (0..count) |i| { + end = if (i + 1 < count) h.list_offs[i + 1] else names_len; + lens[i] = @intCast(end - h.list_offs[i]); + order[i] = i; + } + } + const ctx: SortCtx = .{ .names = h.list_names[0..names_len], .offs = &h.list_offs, .lens = lens }; + std.mem.sort(usize, order[0..count], ctx, SortCtx.lessThan); + for (order[0..count]) |i| { + const name = ctx.nameAt(i); + var child_buf: [rel_capacity + 256]u8 = undefined; + const child_rel = joinRel(&child_buf, dir_rel, name) orelse continue; + const e = h.entryFor(r, child_rel, false) orelse return .overflow; // path table full + st.add(h.entryNode(e), h.list_dirs[i], name); + } + return .ok; +} + +// ---- unit tests ------------------------------------------------------------------- + +const testing = std.testing; + +/// A rig with a fake HOME: every root under one temp dir, never the real +/// ~/.claude or any other live harness root. +const Rig = struct { + dir: testing.TmpDir, + /// Build a fixture process tree under /proc and point + /// `/active` at it. Off by default: a unit test must opt in to + /// reading a process tree, and it is never the live one. + with_proc: bool = false, + /// The pid the daemon believes it is, for the ancestry exclusion. + self_pid: u32 = 4242, + zmx: []const u8 = "zmx", + allow_move: bool = false, + path_buf: [std.fs.max_path_bytes]u8 = undefined, + home: []const u8 = undefined, + h: *Harness = undefined, + harness_mem: Harness = undefined, + + fn start(rig: *Rig) !void { + const io = testing.io; + rig.dir = testing.tmpDir(.{}); + errdefer rig.dir.cleanup(); + const len = try rig.dir.dir.realPath(io, &rig.path_buf); + rig.home = try testing.allocator.dupe(u8, rig.path_buf[0..len]); + var mk: [std.fs.max_path_bytes]u8 = undefined; + // The five roots, pinned to the fake home. + inline for ([_][]const u8{ + ".claude/projects/p1", ".claude/skills/revu", + ".codex/sessions/2026/09/21", + ".omp/agent", ".hermes/logs", + ".dsh/profiles", + }) |sub| { + try Io.Dir.cwd().createDirPath(io, try std.fmt.bufPrint(&mk, "{s}/{s}", .{ rig.home, sub })); + } + var base_buf: [5][std.fs.max_path_bytes]u8 = @splat(@splat(0)); + var bases: [5][]const u8 = @splat(""); + inline for (0..5) |i| { + bases[i] = try std.fmt.bufPrint(&base_buf[i], "{s}/{s}", .{ rig.home, home_dirs[i] }); + } + rig.harness_mem = undefined; + var proc_buf: [std.fs.max_path_bytes]u8 = undefined; + const proc_root: []const u8 = if (rig.with_proc) + try std.fmt.bufPrint(&proc_buf, "{s}/proc", .{rig.home}) + else + ""; + if (rig.with_proc) { + try Io.Dir.cwd().createDirPath(io, proc_root); + try rig.put("proc/stat", "cpu 1 2 3\nbtime 1000000\nprocesses 7\n"); + } + rig.harness_mem.init(.{ + .io = io, + .pid = rig.self_pid, + .bases = bases, + .proc = proc_root, + .home = rig.home, + .zmx = rig.zmx, + .allow_move = rig.allow_move, + }); + rig.h = &rig.harness_mem; + } + + fn end(rig: *Rig) void { + rig.dir.cleanup(); + testing.allocator.free(rig.home); + } + + fn put(rig: *Rig, rel_path: []const u8, bytes: []const u8) !void { + var buf: [std.fs.max_path_bytes]u8 = undefined; + const path = try std.fmt.bufPrint(&buf, "{s}/{s}", .{ rig.home, rel_path }); + if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| { + var dbuf: [std.fs.max_path_bytes]u8 = undefined; + const dir_path = try std.fmt.bufPrint(&dbuf, "{s}/{s}", .{ rig.home, rel_path[0..i] }); + try Io.Dir.cwd().createDirPath(testing.io, dir_path); + } + var file = try Io.Dir.createFileAbsolute(testing.io, path, .{}); + defer file.close(testing.io); + try file.writeStreamingAll(testing.io, bytes); + } + + + fn del(rig: *Rig, rel_path: []const u8) !void { + var buf: [std.fs.max_path_bytes]u8 = undefined; + const path = try std.fmt.bufPrint(&buf, "{s}/{s}", .{ rig.home, rel_path }); + try Io.Dir.deleteFileAbsolute(testing.io, path); + } + + /// lookup of `name` in `dir_node`, answering the child's attr. + fn lookupName(rig: *Rig, tag: u64, dir_node: u64, name: []const u8) Answer { + return handle(rig.h, .{ .tag = tag, .op = .lookup, .node = dir_node, .data = name }); + } + + /// Writes one fake process into the fixture `/proc`: the `stat` + /// line (with the start time in field 22, after a name that holds + /// the spaces and parentheses a real one can), the NUL-separated + /// `cmdline`, and a `cwd` symlink. + fn fakeProc(rig: *Rig, o: struct { + pid: u32, + ppid: u32 = 1, + comm: []const u8 = "x", + starttime: u64 = 5000, + argv: []const u8, + cwd: []const u8 = "", + }) !void { + var rel: [128]u8 = undefined; + var stat_text: [512]u8 = undefined; + var w = Io.Writer.fixed(&stat_text); + try w.print("{d} ({s}) S {d}", .{ o.pid, o.comm, o.ppid }); + for (5..22) |i| try w.print(" {d}", .{i}); + try w.print(" {d} 0 0\n", .{o.starttime}); + try rig.put(try std.fmt.bufPrint(&rel, "proc/{d}/stat", .{o.pid}), w.buffered()); + try rig.put(try std.fmt.bufPrint(&rel, "proc/{d}/cmdline", .{o.pid}), o.argv); + + const target = if (o.cwd.len > 0) o.cwd else rig.home; + var link_buf: [std.fs.max_path_bytes]u8 = undefined; + const link = try std.fmt.bufPrintZ(&link_buf, "{s}/proc/{d}/cwd", .{ rig.home, o.pid }); + Io.Dir.cwd().symLink(testing.io, target, link, .{}) catch {}; + } + + /// Removes a fake process, as an exit would. + fn reapProc(rig: *Rig, pid: u32) !void { + var buf: [std.fs.max_path_bytes]u8 = undefined; + const dir_path = try std.fmt.bufPrint(&buf, "{s}/proc/{d}", .{ rig.home, pid }); + const d = try Io.Dir.openDirAbsolute(testing.io, dir_path, .{ .iterate = true }); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(d, &rb); + var names: [8][64]u8 = undefined; + var lens: [8]usize = @splat(0); + var n: usize = 0; + while (n < names.len) { + const e = (it.next(testing.io) catch break) orelse break; + @memcpy(names[n][0..e.name.len], e.name); + lens[n] = e.name.len; + n += 1; + } + Io.Dir.close(d, testing.io); + for (0..n) |i| { + var one: [std.fs.max_path_bytes]u8 = undefined; + const path = try std.fmt.bufPrint(&one, "{s}/{s}", .{ dir_path, names[i][0..lens[i]] }); + Io.Dir.deleteFileAbsolute(testing.io, path) catch {}; + } + try Io.Dir.cwd().deleteDir(testing.io, dir_path); + } +}; + +const home_dirs = [5][]const u8{ ".claude", ".codex", ".omp", ".hermes", ".dsh" }; + +fn expectNoent(a: Answer) !void { + try testing.expect(a.reply.status == .err); + try testing.expectEqual(E.NOENT, a.reply.errno); +} + +fn expectEperm(a: Answer) !void { + try testing.expect(a.reply.status == .err); + try testing.expectEqual(E.PERM, a.reply.errno); +} + +test "tree: facts at the root" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + const a = rig.lookupName(1, root, "pid"); + try testing.expect(a.reply.status == .ok); + try testing.expect(a.reply.attr.dir == false); + var read_a = handle(rig.h, .{ .tag = 2, .op = .read, .node = a.reply.attr.node, .off = 0, .size = 64 }); + try testing.expectEqualStrings("4242\n", read_a.bytes); + read_a = handle(rig.h, .{ .tag = 3, .op = .read, .node = a.reply.attr.node, .off = 0, .size = 2 }); + try testing.expectEqualStrings("42", read_a.bytes); + const up = rig.lookupName(4, root, "uptime"); + try testing.expect(up.reply.status == .ok); + const up_read = handle(rig.h, .{ .tag = 5, .op = .read, .node = up.reply.attr.node, .off = 0, .size = 64 }); + try testing.expect(up_read.bytes.len > 0 and up_read.bytes[up_read.bytes.len - 1] == '\n'); + // The root lists the facts and the five harness dirs. + const listing = handle(rig.h, .{ .tag = 6, .op = .readdir, .node = root, .off = 0, .size = msize }); + try testing.expect(listing.reply.status == .ok); + for ([_][]const u8{ "README", "pid", "uptime", "claude", "codex", "omp", "hermes", "dsh", "skills" }) |name| { + try testing.expect(stageHas(listing.bytes, name)); + } +} + +test "tree: README is served at the root, whole and read-only" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + + const rd = rig.lookupName(1, root, "README"); + try testing.expect(rd.reply.status == .ok); + try testing.expect(!rd.reply.attr.dir); + try testing.expectEqual(readme_text.len, rd.reply.attr.size); + + // Read it back whole and compare: a truncated or empty README would + // otherwise pass every check above. + const got = handle(rig.h, .{ .tag = 2, .op = .read, .node = rd.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(got.reply.status == .ok); + try testing.expectEqualStrings(readme_text, got.bytes); + + // It is a file, and it is read-only like everything else here. + const as_dir = handle(rig.h, .{ .tag = 3, .op = .readdir, .node = rd.reply.attr.node, .off = 0, .size = msize }); + try testing.expectEqual(E.NOTDIR, as_dir.reply.errno); + const wr = handle(rig.h, .{ .tag = 4, .op = .write, .node = rd.reply.attr.node, .off = 0, .size = 1 }); + try testing.expectEqual(E.PERM, wr.reply.errno); +} + +fn stageHas(staged: []const u8, name: []const u8) bool { + var i: usize = 0; + while (i + 10 <= staged.len) { + const len: usize = staged[i + 9]; + const entry = staged[i + 10 ..][0..len]; + if (std.mem.eql(u8, entry, name)) return true; + i += 10 + len; + } + return false; +} + +test "tree: exclusions never serve a credentials-shaped name" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + // The fixture: a real transcript and a pile of credentials-shaped names + // at several depths, including inside the mirrored subtrees. + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"a\":1}\n"); + try rig.put(".claude/projects/p1/.credentials.json", "STAY OUT\n"); + try rig.put(".claude/projects/p1/auth.json", "STAY OUT\n"); + try rig.put(".claude/projects/p1/settings.json", "STAY OUT\n"); + try rig.put(".claude/projects/p1/token.txt", "STAY OUT\n"); + try rig.put(".claude/projects/p1/api.key", "STAY OUT\n"); + try rig.put(".claude/projects/p1/models.yml", "STAY OUT\n"); + try rig.put(".claude/projects/p1/deep/.env", "STAY OUT\n"); + try rig.put(".claude/history.jsonl", "{\"h\":1}\n"); + try rig.put(".hermes/auth.json", "STAY OUT\n"); + try rig.put(".hermes/.env", "STAY OUT\n"); + try rig.put(".hermes/config.yaml", "STAY OUT\n"); + try rig.put(".hermes/logs/app.log", "log line\n"); + try rig.put(".dsh/.credentials.yaml", "STAY OUT\n"); + try rig.put(".dsh/settings.yaml", "STAY OUT\n"); + + // /claude lists only its mounts; the credentials at ~/.claude root are + // not in the tree at all (no mount serves them). + const claude = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.claude), .off = 0, .size = msize }); + try testing.expect(claude.reply.status == .ok); + for ([_][]const u8{ "projects", "history", "skills" }) |name| try testing.expect(stageHas(claude.bytes, name)); + try testing.expect(!stageHas(claude.bytes, "credentials")); + + // A project's listing shows the transcript and nothing else. + const projects = rig.lookupName(2, topNode(.claude), "projects"); + try testing.expect(projects.reply.status == .ok); + const p1 = rig.lookupName(3, projects.reply.attr.node, "p1"); + try testing.expect(p1.reply.status == .ok); + const listed = handle(rig.h, .{ .tag = 4, .op = .readdir, .node = p1.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(listed.reply.status == .ok); + try testing.expect(stageHas(listed.bytes, "session-x.jsonl")); + for ([_][]const u8{ ".credentials.json", "auth.json", "settings.json", "token.txt", "api.key", "models.yml" }) |name| { + try testing.expect(!stageHas(listed.bytes, name)); + } + // The plain directory that holds a .env is served; the .env is not. + try testing.expect(stageHas(listed.bytes, "deep")); + + // Every credentials-shaped name is unreachable by lookup too. + for ([_][]const u8{ ".credentials.json", "auth.json", "settings.json", "token.txt", "api.key", "models.yml" }) |name| { + try expectNoent(rig.lookupName(5, p1.reply.attr.node, name)); + } + const deep = rig.lookupName(6, p1.reply.attr.node, "deep"); + try testing.expect(deep.reply.status == .ok); + try expectNoent(rig.lookupName(7, deep.reply.attr.node, ".env")); + + // The fully mirrored roots hide theirs as well. + const hermes = rig.lookupName(8, root, "hermes"); + try testing.expect(hermes.reply.status == .ok); + const hlist = handle(rig.h, .{ .tag = 9, .op = .readdir, .node = hermes.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(hlist.bytes, "logs")); + for ([_][]const u8{ "auth.json", ".env", "config.yaml" }) |name| { + try testing.expect(!stageHas(hlist.bytes, name)); + try expectNoent(rig.lookupName(10, hermes.reply.attr.node, name)); + } + const dsh = rig.lookupName(11, root, "dsh"); + try testing.expect(dsh.reply.status == .ok); + const dlist = handle(rig.h, .{ .tag = 12, .op = .readdir, .node = dsh.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(!stageHas(dlist.bytes, ".credentials.yaml")); + try testing.expect(!stageHas(dlist.bytes, "settings.yaml")); + try testing.expect(stageHas(dlist.bytes, "profiles")); + + // The exclusion predicate itself, spelled out. + try testing.expect(excluded(".credentials.json")); + try testing.expect(excluded("AUTH.JSON")); + try testing.expect(excluded("session-token.bin")); + try testing.expect(excluded("id_rsa_backup")); + try testing.expect(excluded("prod.pem")); + try testing.expect(excluded("config.yaml")); + try testing.expect(!excluded("session-x.jsonl")); + try testing.expect(!excluded("SKILL.md")); + try testing.expect(!excluded("logs")); +} + +test "tree: paths compose only from the pinned roots" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + // A slash, a NUL or an overlong name never reaches the filesystem. + try expectNoent(rig.lookupName(1, topNode(.claude), "projects/../p1")); + try expectNoent(rig.lookupName(2, topNode(.claude), "projects/\x00")); + // A symlink inside a mirror is not served, whatever it points at. + var buf: [std.fs.max_path_bytes]u8 = undefined; + const link = try std.fmt.bufPrintZ(&buf, "{s}/.claude/projects/p1/escape", .{rig.home}); + try Io.Dir.cwd().symLink(testing.io, rig.home, link, .{}); + const projects = rig.lookupName(3, topNode(.claude), "projects"); + const p1 = rig.lookupName(4, projects.reply.attr.node, "p1"); + try expectNoent(rig.lookupName(5, p1.reply.attr.node, "escape")); + // ".." from a mirrored file lands on its canonical parent, and walking + // ".." repeatedly terminates at the root. + const session = rig.lookupName(6, p1.reply.attr.node, "session-x.jsonl"); + try testing.expect(session.reply.status == .ok); + var cur = session.reply.attr.node; + var hops: usize = 0; + while (hops < 8) : (hops += 1) { + const up = rig.lookupName(7, cur, ".."); + try testing.expect(up.reply.status == .ok); + if (up.reply.attr.node == root) break; + cur = up.reply.attr.node; + } + try testing.expect(cur == root or hops < 8); + // The facts refuse lookups with NOTDIR. + try testing.expect(rig.lookupName(8, topNode(.pid), "x").reply.errno == E.NOTDIR); +} + +test "tree: a file's qid path survives release, and its version tracks content" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const dir = rig.lookupName(2, projects.reply.attr.node, "p1"); + const first = rig.lookupName(3, dir.reply.attr.node, "session-x.jsonl"); + try testing.expect(first.reply.status == .ok); + const qid_path = first.reply.attr.path.?; + const version = first.reply.attr.version; + + // Give every reference back, so the table slot is free to recycle. + for ([_]u64{ first.reply.attr.node, dir.reply.attr.node, projects.reply.attr.node }) |n| { + _ = handle(rig.h, .{ .tag = 4, .op = .release, .node = n }); + } + + // Walk to it again from scratch. intro(5): two files are the same file if + // and only if their qids are the same — so the same unchanged file must + // come back with the same qid path, whatever the node id does. Before the + // qid path came from the file itself, the recycled slot's new generation + // made this a different file on every walk. + const p2 = rig.lookupName(5, topNode(.claude), "projects"); + const d2 = rig.lookupName(6, p2.reply.attr.node, "p1"); + const again = rig.lookupName(7, d2.reply.attr.node, "session-x.jsonl"); + try testing.expect(again.reply.status == .ok); + try testing.expectEqual(qid_path, again.reply.attr.path.?); + try testing.expectEqual(version, again.reply.attr.version); + + // Two different files never share a qid path. + try rig.put(".claude/projects/p1/session-y.jsonl", "{}\n"); + const other = rig.lookupName(8, d2.reply.attr.node, "session-y.jsonl"); + try testing.expect(other.reply.attr.path.? != qid_path); + + // And a change to the content moves the version, which is what tells a + // caching client the transcript it is holding has grown. + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n{\"more\":1}\n"); + const grown = rig.lookupName(9, d2.reply.attr.node, "session-x.jsonl"); + try testing.expectEqual(qid_path, grown.reply.attr.path.?); + try testing.expect(grown.reply.attr.version != version); +} + +test "tree: node ids — same file equal, distinct files differ, ids recycle" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + try rig.put(".claude/projects/p1/session-y.jsonl", "{}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const a1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + // Two walks of the same file: the same node id (the dedupe table). + const x1 = rig.lookupName(3, a1.reply.attr.node, "session-x.jsonl"); + const x2 = rig.lookupName(4, a1.reply.attr.node, "session-x.jsonl"); + try testing.expect(x1.reply.status == .ok); + try testing.expectEqual(x1.reply.attr.node, x2.reply.attr.node); + const y1 = rig.lookupName(5, a1.reply.attr.node, "session-y.jsonl"); + try testing.expect(y1.reply.attr.node != x1.reply.attr.node); + // The union skills view and the harness's own skills share identity. + const via_harness = rig.lookupName(6, topNode(.claude), "skills"); + const via_union = rig.lookupName(7, topNode(.skills), "claude"); + try testing.expectEqual(via_harness.reply.attr.node, via_union.reply.attr.node); + // References are paid back by release: both slots go, ids recycle. + for ([_]u64{ x1.reply.attr.node, x2.reply.attr.node, y1.reply.attr.node, a1.reply.attr.node }) |n| { + const rel = handle(rig.h, .{ .tag = 8, .op = .release, .node = n }); + try testing.expect(rel.reply.status == .ok); + } + const x3 = rig.lookupName(9, projects.reply.attr.node, "p1"); + const x4 = rig.lookupName(10, x3.reply.attr.node, "session-x.jsonl"); + // The entry was freed and re-created: a fresh generation, a new id. + try testing.expect(x4.reply.attr.node != x1.reply.attr.node); + // Node packing round-trips through the struct. + const n: Node = .{ .idx = 3, .kind = 1, .serial = 0x1234_5678_9abc }; + const bits: u64 = @bitCast(n); + const back: Node = @bitCast(bits); + try testing.expect(back.idx == 3 and back.kind == 1 and back.serial == 0x1234_5678_9abc); +} + +test "tree: a file that vanishes between lookup and read answers ENOENT" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"a\":1}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); + try testing.expect(session.reply.status == .ok); + // The transcript reads back whole, fresh from disk. + const whole = handle(rig.h, .{ .tag = 4, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 1024 }); + try testing.expectEqualStrings("{\"a\":1}\n", whole.bytes); + // It vanishes: lookup, getattr and read all answer ENOENT cleanly. + try rig.del(".claude/projects/p1/session-x.jsonl"); + try expectNoent(rig.lookupName(5, p1.reply.attr.node, "session-x.jsonl")); + const gone = handle(rig.h, .{ .tag = 6, .op = .getattr, .node = session.reply.attr.node }); + try expectNoent(gone); + const read_gone = handle(rig.h, .{ .tag = 7, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 64 }); + try expectNoent(read_gone); +} + +test "tree: writes and setattrs answer EPERM" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); + try testing.expect(session.reply.status == .ok); + const write = handle(rig.h, .{ .tag = 4, .op = .write, .node = session.reply.attr.node, .data = "x" }); + try expectEperm(write); + const set = handle(rig.h, .{ .tag = 5, .op = .setattr, .node = session.reply.attr.node, .set = .{ .mtime = true }, .mtime = 1 }); + try expectEperm(set); + // An open for write is refused at open time. + const wopen = handle(rig.h, .{ .tag = 6, .op = .open, .node = session.reply.attr.node, .omode = cloud9.owrite }); + try expectEperm(wopen); + const ropen = handle(rig.h, .{ .tag = 7, .op = .open, .node = session.reply.attr.node, .omode = cloud9.oread }); + try testing.expect(ropen.reply.status == .ok); +} + +test "tree: a transcript written while serving is visible at once" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"n\":1}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + // The harness appends; the next read sees it — no cache in between. + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"n\":1}\n{\"n\":2}\n"); + const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); + const fresh = handle(rig.h, .{ .tag = 4, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 1024 }); + try testing.expectEqualStrings("{\"n\":1}\n{\"n\":2}\n", fresh.bytes); + // A brand-new session file appears on the next listing. + try rig.put(".claude/projects/p1/session-new.jsonl", "{\"n\":9}\n"); + const listed = handle(rig.h, .{ .tag = 5, .op = .readdir, .node = p1.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(listed.bytes, "session-new.jsonl")); +} + +comptime { + // Every static mount target must survive the exclusion rules; the + // daemon's own skeleton may never be filtered out from under it. + for (named_mounts) |m| { + if (excludedPath(m.mount.rel)) @compileError("a 9agents mount target is excluded by name"); + } +} + +fn stageCount(staged: []const u8) usize { + var i: usize = 0; + var n: usize = 0; + while (i + 10 <= staged.len) : (n += 1) i += 10 + @as(usize, staged[i + 9]); + return n; +} + +test "tree: a file directly under a mirror root reads back" { + // Regression: joining a child onto an empty relative path returned an + // uncopied scratch slice, so every file at the top of a mirror root + // (/hermes/, /dsh/) listed but resolved to garbage bytes. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".hermes/note.txt", "hello-hermes\n"); + const listing = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.hermes), .off = 0, .size = msize }); + try testing.expect(stageHas(listing.bytes, "note.txt")); + const note = rig.lookupName(2, topNode(.hermes), "note.txt"); + try testing.expect(note.reply.status == .ok); + const bytes = handle(rig.h, .{ .tag = 3, .op = .read, .node = note.reply.attr.node, .off = 0, .size = 64 }); + try testing.expectEqualStrings("hello-hermes\n", bytes.bytes); +} + +test "tree: a name swapped for a symlink under an open handle serves nothing" { + // Regression: the read path composed the whole path and opened it in + // one call, following symlinks. A name replaced between the walk and + // the read handed the client bytes from outside every pinned root. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put("outside.txt", "OUTSIDE-THE-ROOTS\n"); // beside the roots, not in one + try rig.put(".claude/projects/p1/swap.txt", "safe\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + const swap = rig.lookupName(3, p1.reply.attr.node, "swap.txt"); + try testing.expect(swap.reply.status == .ok); + const opened = handle(rig.h, .{ .tag = 4, .op = .open, .node = swap.reply.attr.node, .omode = cloud9.oread }); + try testing.expect(opened.reply.status == .ok); + // The file becomes a symlink out of the tree while the handle is open. + var link_buf: [std.fs.max_path_bytes]u8 = undefined; + var target_buf: [std.fs.max_path_bytes]u8 = undefined; + const link = try std.fmt.bufPrintZ(&link_buf, "{s}/.claude/projects/p1/swap.txt", .{rig.home}); + const target = try std.fmt.bufPrint(&target_buf, "{s}/outside.txt", .{rig.home}); + try rig.del(".claude/projects/p1/swap.txt"); + try Io.Dir.cwd().symLink(testing.io, target, link, .{}); + const after = handle(rig.h, .{ .tag = 5, .op = .read, .node = swap.reply.attr.node, .off = 0, .size = 64 }); + try expectNoent(after); + try testing.expectEqual(@as(usize, 0), after.bytes.len); +} + +test "tree: a listing past the cap fails loudly instead of truncating" { + // Regression: a directory with more entries (or longer names) than the + // comptime caps was served short, and a short listing is + // indistinguishable from a small directory. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + var name_buf: [64]u8 = undefined; + for (0..list_capacity + 1) |i| { + try rig.put(try std.fmt.bufPrint(&name_buf, ".dsh/f{d:0>5}", .{i}), ""); + } + const listed = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.dsh), .off = 0, .size = msize }); + try testing.expect(listed.reply.status == .err); + try testing.expectEqual(E.NFILE, listed.reply.errno); +} + +test "tree: a listing spanning several reads loses no entry" { + // The staging buffer fills long before a big directory ends. Staging + // must stop at the first record that does not fit: dropping it and + // packing a shorter one behind it would lose that entry, because the + // next read's offset counts the records already delivered. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + const count = 200; + var name_buf: [320]u8 = undefined; + for (0..count) |i| { + const name = try std.fmt.bufPrint(&name_buf, ".dsh/{d:0>3}{s}", .{ i, "n" ** 200 }); + try rig.put(name, ""); + } + var seen: [count]bool = @splat(false); + var off: u64 = 0; + var reads: usize = 0; + while (reads < 16) : (reads += 1) { + const page = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.dsh), .off = off, .size = msize }); + try testing.expect(page.reply.status == .ok); + if (page.bytes.len == 0) break; + var i: usize = 0; + while (i + 10 <= page.bytes.len) { + const len: usize = page.bytes[i + 9]; + const name = page.bytes[i + 10 ..][0..len]; + i += 10 + len; + // The fixture root also holds `profiles`, created by the rig. + const idx = std.fmt.parseInt(usize, name[0..@min(3, name.len)], 10) catch continue; + try testing.expect(!seen[idx]); // never served twice + seen[idx] = true; + } + off += stageCount(page.bytes); + } + try testing.expect(reads > 1); // the listing really did span several reads + for (seen) |s| try testing.expect(s); +} + +test { + _ = active; // the derived view's own tests run with the tree's +} + +// ---- /active: the derived view ------------------------------------------------ + +/// Walks `/active` down to one agent's directory, answering its node. +fn activeEntryNode(rig: *Rig, harness: []const u8, pid: []const u8) !u64 { + const act = rig.lookupName(90, root, "active"); + try testing.expect(act.reply.status == .ok); + // A listing is what rescans, so it comes before the walk. + _ = handle(rig.h, .{ .tag = 91, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + const h_dir = rig.lookupName(92, act.reply.attr.node, harness); + try testing.expect(h_dir.reply.status == .ok); + const entry = rig.lookupName(93, h_dir.reply.attr.node, pid); + try testing.expect(entry.reply.status == .ok); + return entry.reply.attr.node; +} + +fn activeField(rig: *Rig, entry: u64, name: []const u8) ![]const u8 { + const f = rig.lookupName(94, entry, name); + try testing.expect(f.reply.status == .ok); + const r = handle(rig.h, .{ .tag = 95, .op = .read, .node = f.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(r.reply.status == .ok); + return r.bytes; +} + +test "active: a fixture process tree lists by harness and by pid" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + + // A Claude Code session, resolved the way the harness publishes it. + try rig.fakeProc(.{ .pid = 1001, .comm = "claude", .starttime = 5000, .argv = "claude\x00--print\x00" }); + var rec: [512]u8 = undefined; + try rig.put(".claude/sessions/1001.json", try std.fmt.bufPrint(&rec, + \\{{"pid":1001,"sessionId":"sess-abc","cwd":"{s}", + \\ "name":"fixture-one","status":"busy","procStart":"5000"}} + , .{rig.home})); + var slug_buf: [512]u8 = undefined; + const slug = active.slugOf(.claude, rig.home, rig.home, &slug_buf).?; + var tr: [640]u8 = undefined; + try rig.put( + try std.fmt.bufPrint(&tr, ".claude/projects/{s}/sess-abc.jsonl", .{slug}), + "{\"type\":\"summary\",\"aiTitle\":\"Fixture Title\"}\n", + ); + + const entry = try activeEntryNode(&rig, "claude", "1001"); + try testing.expectEqualStrings("1001\n", try activeField(&rig, entry, "pid")); + try testing.expectEqualStrings("sess-abc\n", try activeField(&rig, entry, "session")); + try testing.expectEqualStrings("registry\n", try activeField(&rig, entry, "via")); + try testing.expectEqualStrings("fixture-one\n", try activeField(&rig, entry, "name")); + try testing.expectEqualStrings("busy\n", try activeField(&rig, entry, "status")); + try testing.expectEqualStrings("Fixture Title\n", try activeField(&rig, entry, "title")); + // started is the boot time plus the process's own, in seconds. + try testing.expectEqualStrings("1000050\n", try activeField(&rig, entry, "started")); + // The transcript is served as the file it is, not as a field. + const t = rig.lookupName(96, entry, "transcript"); + try testing.expect(t.reply.status == .ok); + const bytes = handle(rig.h, .{ .tag = 97, .op = .read, .node = t.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(std.mem.indexOf(u8, bytes.bytes, "Fixture Title") != null); + // A field this harness does not publish is absent, not blank. + try expectNoent(rig.lookupName(98, entry, "model")); +} + +test "active: a daemon or helper is never listed as an agent" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + // The shapes actually running on this machine. + try rig.fakeProc(.{ .pid = 1010, .comm = "python3", .argv = "python3\x00-m\x00hermes_cli.main\x00gateway\x00run\x00" }); + try rig.fakeProc(.{ .pid = 1011, .comm = "omp", .argv = "omp\x00__omp_worker_daemon_broker\x00" }); + try rig.fakeProc(.{ .pid = 1012, .comm = "claude", .argv = "claude\x00mcp-server\x00" }); + // ...and one real session, so an empty answer cannot pass by default. + try rig.fakeProc(.{ .pid = 1013, .comm = "claude", .argv = "claude\x00" }); + + const act = rig.lookupName(1, root, "active"); + const listing = handle(rig.h, .{ .tag = 2, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(listing.bytes, "claude")); + try testing.expect(!stageHas(listing.bytes, "hermes")); + try testing.expect(!stageHas(listing.bytes, "omp")); + const claude = rig.lookupName(3, act.reply.attr.node, "claude"); + const pids = handle(rig.h, .{ .tag = 4, .op = .readdir, .node = claude.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(pids.bytes, "1013")); + try testing.expect(!stageHas(pids.bytes, "1012")); // the mcp server +} + +test "active: a pid reused by another process stops resolving" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + try rig.fakeProc(.{ .pid = 1020, .comm = "claude", .starttime = 5000, .argv = "claude\x00" }); + const entry = try activeEntryNode(&rig, "claude", "1020"); + try testing.expect(handle(rig.h, .{ .tag = 5, .op = .getattr, .node = entry }).reply.status == .ok); + + // The same pid, a different process: only the start time says so. + try rig.fakeProc(.{ .pid = 1020, .comm = "claude", .starttime = 9999, .argv = "claude\x00" }); + const act = rig.lookupName(6, root, "active"); + _ = handle(rig.h, .{ .tag = 7, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + try expectNoent(handle(rig.h, .{ .tag = 8, .op = .getattr, .node = entry })); + // The pid is still there — as the new process, under a new node. + const fresh = try activeEntryNode(&rig, "claude", "1020"); + try testing.expect(fresh != entry); +} + +test "active: a process that exits leaves the tree" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + try rig.fakeProc(.{ .pid = 1030, .comm = "claude", .argv = "claude\x00" }); + const entry = try activeEntryNode(&rig, "claude", "1030"); + try testing.expect(handle(rig.h, .{ .tag = 9, .op = .getattr, .node = entry }).reply.status == .ok); + + try rig.reapProc(1030); + const act = rig.lookupName(10, root, "active"); + const listing = handle(rig.h, .{ .tag = 11, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(!stageHas(listing.bytes, "claude")); + try expectNoent(handle(rig.h, .{ .tag = 12, .op = .getattr, .node = entry })); + try expectNoent(rig.lookupName(13, act.reply.attr.node, "claude")); +} + +test "active: the daemon never lists the process tree it lives in" { + // 1041 is the daemon, its parent 1040 is a harness: showing it would + // let a client act on the tree serving it. + var rig: Rig = .{ .dir = undefined, .with_proc = true, .self_pid = 1041 }; + try rig.start(); + defer rig.end(); + try rig.fakeProc(.{ .pid = 1040, .comm = "claude", .argv = "claude\x00" }); + try rig.fakeProc(.{ .pid = 1041, .comm = "9agents", .ppid = 1040, .argv = "9agents\x00" }); + try rig.fakeProc(.{ .pid = 1042, .comm = "claude", .argv = "claude\x00" }); + + const act = rig.lookupName(14, root, "active"); + const claude = rig.lookupName(15, act.reply.attr.node, "claude"); + try testing.expect(claude.reply.status == .ok); + const pids = handle(rig.h, .{ .tag = 16, .op = .readdir, .node = claude.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(pids.bytes, "1042")); // an unrelated session lists + try testing.expect(!stageHas(pids.bytes, "1040")); // its own parent does not +} diff --git a/9agents/test/e2e.sh b/9agents/test/e2e.sh new file mode 100755 index 0000000..fa41bef --- /dev/null +++ b/9agents/test/e2e.sh @@ -0,0 +1,358 @@ +#!/usr/bin/env bash +# End-to-end suite for 9agents: the read-only, fresh-from-disk 9P view of +# every harness's state. +# +# Usage: bash 9agents/test/e2e.sh <9agents> [<9ns>] (zig build 9agents-itest) +# +# Part A always runs, entirely on fixtures: a fake home with fixture roots +# pinned by --root, a scratch XDG_RUNTIME_DIR registry, plan9port's 9p as +# the client. It proves: the posted name is listed, the roots walk and +# read, a transcript comes back byte-identical (cmp), credentials-shaped +# files are unreachable, a file appended after the daemon started is +# visible at once, writes answer EPERM, and --unix/--fd listen forms work. +# +# Part B (the money shot) runs against the REAL roots and the REAL +# registry only when the `agents` name is free, and only reads: the posted +# name in /run/user//9p, 9p walks of the live ~/.claude, a real +# transcript byte-identical through the tree, the 9ns --mntgen mount of +# /mnt/9p/agents, and the live ~/.claude/.credentials.json unreachable. +# It is skipped (not failed) when a live daemon already owns the name. +# +# Exit 0 on success (or when the machine cannot run a part), 1 on failure. +set -u + +H9=$(realpath "${1:?path to 9agents}") +HARNESS_TMP_XDG="${XDG_RUNTIME_DIR:-}" +[ $# -ge 2 ] && [ -n "$2" ] && NS=$(realpath "$2") +P9P=/usr/lib/plan9/bin/9p +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9agents.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" + # part B posts into the *real* registry under a scratch name; a crash + # between the post and the unpost would otherwise leave a socket there. + [ -n "${REAL_SOCKET:-}" ] && rm -f "$REAL_SOCKET" + return 0 +} +trap cleanup EXIT + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { # name expected actual + if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi +} +expect_contains() { # name needle haystack + case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac +} +expect_missing() { # name needle haystack + case "$3" in *"$2"*) fail "$1" "found: $(printf %q "$2")" "in: $(printf %q "$3")" ;; *) pass "$1" ;; esac +} + +if [ ! -x "$P9P" ]; then + echo "SKIP: plan9port 9p not at $P9P"; exit 0 +fi +if ! unshare -Urm true 2>/dev/null; then + echo "SKIP: unprivileged user namespaces unavailable"; exit 0 +fi + +start_daemon() { # args... -> sets DAEMON_PID, logs to $TMP/daemon.log + "$H9" "$@" >"$TMP/daemon.log" 2>&1 & + DAEMON_PID=$! + PIDS+=("$DAEMON_PID") +} +wait_posted() { # socket path + for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done + return 1 +} +p9() { "$P9P" -a "unix!$1" "${@:2}"; } + +# ============================================================================ +echo "# part A: fixture roots, scratch registry" +# ============================================================================ +export XDG_RUNTIME_DIR="$TMP" +REG="$TMP/9p" +FX="$TMP/home" +mkdir -p "$FX" + +# The fixture home: five roots, a real-shaped claude project with a +# transcript, credentials-shaped files at several depths, codex sessions, +# an omp agent dir, hermes logs and dsh profiles. +mkfile() { mkdir -p "$(dirname "$1")"; printf '%s' "$2" > "$1"; } +mkfile "$FX/claude/projects/-tmp-proj/session-abc.jsonl" '{"type":"user","message":"first line"} +{"type":"assistant","message":"second line"} +' +mkfile "$FX/claude/projects/-tmp-proj/.credentials.json" 'STAY OUT' +mkfile "$FX/claude/projects/-tmp-proj/auth.json" 'STAY OUT' +mkfile "$FX/claude/projects/-tmp-proj/token.bin" 'STAY OUT' +mkfile "$FX/claude/history.jsonl" '{"display":"claude prompt one"} +{"display":"claude prompt two"} +' +mkdir -p "$FX/claude/skills/revu" +mkfile "$FX/claude/skills/revu/SKILL.md" '# revu skill' +mkfile "$FX/codex/sessions/2026/09/21/rollout-x.jsonl" '{"session":"codex one"} +' +mkfile "$FX/codex/session_index.jsonl" '{"id":"x"} +' +mkfile "$FX/codex/history.jsonl" '{"text":"codex prompt"} +' +mkfile "$FX/omp/agent/history.db" 'fake-history-db' +mkfile "$FX/omp/agent/config.yml" 'STAY OUT' +mkfile "$FX/hermes/auth.json" 'STAY OUT' +mkfile "$FX/hermes/logs/app.log" 'hermes log line +' +mkfile "$FX/hermes/state.db" 'fake-hermes-state' +mkfile "$FX/dsh/profiles/p1.yaml" 'name: p1' +mkfile "$FX/dsh/.credentials.yaml" 'STAY OUT' + +start_daemon --root "claude=$FX/claude" --root "codex=$FX/codex" \ + --root "omp=$FX/omp" --root "hermes=$FX/hermes" --root "dsh=$FX/dsh" \ + --name agents +wait_posted "$REG/agents" || { fail "the daemon posts as agents" "$(cat "$TMP/daemon.log")"; exit 1; } + +# 1. The posted name is listed. +expect_contains "posted name listed in the registry" agents "$(ls "$REG")" + +# 2. The roots walk; a real transcript reads back byte-identical. +expect_contains "ls / shows the roots" claude "$(p9 "$REG/agents" ls /)" +expect_contains "ls / shows codex" codex "$(p9 "$REG/agents" ls /)" +expect_contains "ls / shows skills" skills "$(p9 "$REG/agents" ls /)" +p9 "$REG/agents" read /claude/projects/-tmp-proj/session-abc.jsonl > "$TMP/via-9p.jsonl" 2>"$TMP/read.err" \ + || fail "transcript read through the tree" "$(cat "$TMP/read.err")" +cmp -s "$TMP/via-9p.jsonl" "$FX/claude/projects/-tmp-proj/session-abc.jsonl" \ + && pass "transcript byte-identical through the tree (cmp)" \ + || fail "transcript byte-identical through the tree (cmp)" "differs" +expect_eq "history file reads" "$(cat "$FX/claude/history.jsonl")" "$(p9 "$REG/agents" read /claude/history)" +expect_eq "skills union mirrors the harness tree" "$(cat "$FX/claude/skills/revu/SKILL.md")" \ + "$(p9 "$REG/agents" read /skills/claude/revu/SKILL.md)" + +# 3. A credentials-shaped file is unreachable anywhere in the tree. +PROJ_LS=$(p9 "$REG/agents" ls /claude/projects/-tmp-proj) +for bad in .credentials.json auth.json token.bin; do + expect_missing "credentials-shaped $bad not listed" "$bad" "$PROJ_LS" + p9 "$REG/agents" read "/claude/projects/-tmp-proj/$bad" >/dev/null 2>&1 \ + && fail "credentials-shaped $bad unreachable" "read succeeded" \ + || pass "credentials-shaped $bad unreachable" +done +expect_missing "hermes auth.json not listed" auth.json "$(p9 "$REG/agents" ls /hermes)" +expect_missing "omp config.yml not listed" config.yml "$(p9 "$REG/agents" ls /omp)" +expect_missing "dsh .credentials.yaml not listed" credentials.yaml "$(p9 "$REG/agents" ls /dsh)" +sleep 0.3 +expect_contains "hermes logs still served" logs "$(p9 "$REG/agents" ls /hermes)" + +# 3b. A file at the very top of a mirror root: its relative path is the +# bare name, the one case a join onto an empty directory path gets wrong. +expect_contains "a file at the top of a mirror root is listed" state.db "$(p9 "$REG/agents" ls /hermes)" +expect_eq "a file at the top of a mirror root reads back" "$(cat "$FX/hermes/state.db")" \ + "$(p9 "$REG/agents" read /hermes/state.db)" + +# 3c. A directory bigger than the listing caps fails loudly. A short +# listing is indistinguishable from a small directory, so the daemon must +# never answer one: the read errors and the client sees it. +mkdir -p "$FX/dsh/wide" +seq 1 1100 | while read -r i; do : > "$FX/dsh/wide/f$(printf %05d "$i")"; done +if p9 "$REG/agents" ls /dsh/wide > "$TMP/wide.out" 2>"$TMP/wide.err"; then + fail "an over-cap directory fails instead of truncating" "listed $(wc -l < "$TMP/wide.out") entries" +else + pass "an over-cap directory fails instead of truncating ($(head -c 80 "$TMP/wide.err"))" +fi +rm -rf "$FX/dsh/wide" + +# 4. Live visibility: a file appended after the daemon started. +printf '{"type":"assistant","message":"third line"}\n' >> "$FX/claude/projects/-tmp-proj/session-abc.jsonl" +expect_eq "append after start is visible immediately" \ + "$(cat "$FX/claude/projects/-tmp-proj/session-abc.jsonl")" \ + "$(p9 "$REG/agents" read /claude/projects/-tmp-proj/session-abc.jsonl)" +mkdir -p "$FX/claude/projects/-tmp-proj2" +mkfile "$FX/claude/projects/-tmp-proj2/session-new.jsonl" '{"new":true} +' +expect_contains "new session dir appears at once" -tmp-proj2 "$(p9 "$REG/agents" ls /claude/projects)" + +# 5. The facts and the write refusal. +DAEMON_PID_TEXT=$(p9 "$REG/agents" read /pid) +expect_eq "pid fact answers the daemon's pid" "$DAEMON_PID" "$DAEMON_PID_TEXT" +[ -n "$(p9 "$REG/agents" read /uptime)" ] && pass "uptime fact answers" || fail "uptime fact answers" "empty" +printf 'x' | p9 "$REG/agents" write /pid >/dev/null 2>&1 \ + && fail "write answers EPERM" "write succeeded" \ + || pass "write answers EPERM" + +# 6. The --unix listen form beside the post. +"$H9" --unix "$TMP/plain.sock" --no-post --root "claude=$FX/claude" >"$TMP/d2.log" 2>&1 & +PIDS+=($!) +for _ in $(seq 1 100); do [ -S "$TMP/plain.sock" ] && break; sleep 0.05; done +expect_eq "--unix listen form serves" "$(cat "$FX/claude/history.jsonl")" "$(p9 "$TMP/plain.sock" read /claude/history)" + +# 7. The --fd listen form: one 9P session over a connected stream fd +# (the 9ns --spawn / socket-activation shape), driven through a +# socketpair: the daemon sees EOF when both ends close and exits. +if command -v python3 >/dev/null; then + python3 - "$H9" "$FX" <<'EOF' +import os, socket, subprocess, sys +h9, fx = sys.argv[1], sys.argv[2] +a, b = socket.socketpair() +pid = os.fork() +if pid == 0: + a.close() + os.dup2(b.fileno(), 7) + os.execv(h9, [h9, "--fd", "7", "--root", f"claude={fx}/claude"]) +b.close() +a.close() +os.waitpid(pid, 0) +EOF + pass "--fd listen form runs a session and exits on hangup" +else + echo "skip - --fd form: python3 not available" +fi + +kill "$DAEMON_PID" 2>/dev/null +wait "$DAEMON_PID" 2>/dev/null +sleep 0.2 +[ -S "$REG/agents" ] && fail "SIGTERM unposts the name" "socket still there" || pass "SIGTERM unposts the name" + +# ============================================================================ +echo "# part C: /active, the derived view" +# ============================================================================ +# A fake agent: a process whose argv[0] basename is a harness name, with a +# session record Claude Code's own shape. The proc root is the real /proc +# (the fixture seam is unit-tested); nothing real is ever killed, because +# the only process this part touches is the sleep it started itself. +ACT="$TMP/act" +mkdir -p "$ACT/bin" "$ACT/home/.claude/sessions" "$ACT/home/.claude/projects" "$ACT/rt/9p/zmx" +cp /bin/sleep "$ACT/bin/claude" +# A zmx that records how it was called and posts the name, as the real one +# does; a move must never be tested against the user's live sessions. +cat > "$ACT/bin/zmx" < "$ACT/zmx-argv" +: > "$ACT/rt/9p/zmx/\$2" +exit 0 +ZEOF +chmod +x "$ACT/bin/zmx" + +"$ACT/bin/claude" 600 & +AGENT=$! +PIDS+=("$AGENT") +sleep 0.3 +# Field 22 of /proc//stat, counted from the last ')' — the executable +# name can hold spaces and parentheses, so the line is never just split. +AGENT_START=$(sed 's/.*) //' "/proc/$AGENT/stat" | awk '{print $20}') +AGENT_CWD=$(readlink "/proc/$AGENT/cwd") +printf '{"pid":%d,"sessionId":"sess-e2e","cwd":"%s","name":"e2e-agent","status":"idle","procStart":"%s"}\n' \ + "$AGENT" "$AGENT_CWD" "$AGENT_START" > "$ACT/home/.claude/sessions/$AGENT.json" + +act_roots() { + echo --root "claude=$ACT/home/.claude" --root "codex=$ACT/home/.codex" \ + --root "omp=$ACT/home/.omp" --root "hermes=$ACT/home/.hermes" --root "dsh=$ACT/home/.dsh" +} + +# First: the read-only default. No --allow-move, so zmx cannot be written. +XDG_RUNTIME_DIR="$ACT/rt" "$H9" --no-post --unix "$ACT/ro.sock" $(act_roots) >"$ACT/ro.log" 2>&1 & +PIDS+=("$!") +wait_posted "$ACT/ro.sock" || { fail "read-only daemon listens" "$(cat "$ACT/ro.log")"; } +expect_contains "the agent lists under its harness" "$AGENT" "$(p9 "$ACT/ro.sock" ls /active/claude)" +expect_eq "its session comes from the harness's own record" "sess-e2e" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/session")" +expect_eq "the route taken is named" "registry" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/via")" +expect_eq "the harness's own name is served" "e2e-agent" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/name")" +expect_eq "the harness's own status is served" "idle" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/status")" +if echo "nope" | p9 "$ACT/ro.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null; then + fail "without --allow-move the tree stays read-only" "the write was accepted" +else + pass "without --allow-move the tree stays read-only" +fi +expect_eq "a refused move leaves the agent running" "yes" "$([ -d "/proc/$AGENT" ] && echo yes)" + +# Now a daemon that allows moves. +XDG_RUNTIME_DIR="$ACT/rt" "$H9" --no-post --unix "$ACT/rw.sock" $(act_roots) \ + --allow-move --zmx "$ACT/bin/zmx" >"$ACT/rw.log" 2>&1 & +PIDS+=("$!") +wait_posted "$ACT/rw.sock" || { fail "move-allowing daemon listens" "$(cat "$ACT/rw.log")"; } + +# Every refusal must come before anything is destroyed. +for bad in "has space" "../escape" "semi;colon"; do + echo "$bad" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null + if [ $? -eq 0 ]; then fail "an illegal zmx name is refused ($bad)"; else pass "an illegal zmx name is refused ($bad)"; fi +done +: > "$ACT/rt/9p/zmx/occupied" +echo "occupied" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null \ + && fail "a name a live session holds is refused" || pass "a name a live session holds is refused" +expect_eq "no refusal killed the agent" "yes" "$([ -d "/proc/$AGENT" ] && echo yes)" + +# The move itself. +if echo "e2e-moved" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>"$ACT/move.err"; then + pass "a legal name moves the agent" +else + fail "a legal name moves the agent" "$(cat "$ACT/move.err")" +fi +sleep 0.4 +expect_eq "the agent it replaced is gone" "gone" "$([ -d "/proc/$AGENT" ] || echo gone)" +# The command is fixed by the harness: the client supplied only the name. +expect_eq "zmx ran the harness with its session resumed" \ + "run e2e-moved -d claude --resume sess-e2e" "$(cat "$ACT/zmx-argv")" +expect_missing "no client byte reached the command" "e2e-moved -d claude --resume sess-e2e;" "$(cat "$ACT/zmx-argv")" + +# ============================================================================ +echo "# part B: the real roots, the real registry (read-only)" +# ============================================================================ +export XDG_RUNTIME_DIR="${HARNESS_TMP_XDG:-/run/user/$(id -u)}" +REAL_REG="$XDG_RUNTIME_DIR/9p" +# A scratch name, not `agents`: the whole point of part B is to read the real +# roots through the real registry, and that has nothing to do with which name +# the tree is posted under. Taking `agents` would mean this suite could only +# run while the machine's own 9agents.service was stopped — so it would either +# be skipped on any machine that actually uses the daemon, or fight it. +REAL_NAME="agents-itest-$$" +REAL_SOCKET="$REAL_REG/$REAL_NAME" +if [ -e "$REAL_SOCKET" ]; then + # $$ collided with a leftover socket from a crashed run of this suite. + echo "SKIP: scratch name $REAL_NAME is already taken" +else + HOME_DIR=$(getent passwd "$(id -u)" | cut -d: -f6) + start_daemon --name "$REAL_NAME" + if wait_posted "$REAL_SOCKET"; then + expect_contains "posted name listed in the real registry" "$REAL_NAME" "$(ls "$REAL_REG")" + expect_contains "ls / shows the claude root" claude "$(p9 "$REAL_SOCKET" ls /)" + # A real transcript, byte-identical through the tree. + REAL_T=$(ls "$HOME_DIR/.claude/projects"/*/*.jsonl 2>/dev/null | head -n 1 || true) + if [ -n "$REAL_T" ]; then + REL="${REAL_T#"$HOME_DIR/.claude/projects/"}" + p9 "$REAL_SOCKET" read "/claude/projects/$REL" > "$TMP/real-via-9p" 2>/dev/null \ + && cmp -s "$TMP/real-via-9p" "$REAL_T" \ + && pass "real transcript byte-identical through the tree" \ + || fail "real transcript byte-identical through the tree" "$REAL_T" + else + echo "skip - no real claude transcript found" + fi + # The credentials at ~/.claude are unreachable: no mount serves the + # root dir, and the exclusion rule holds everywhere else. + p9 "$REAL_SOCKET" read /claude/.credentials.json >/dev/null 2>&1 \ + && fail "live .credentials.json unreachable" "read succeeded" \ + || pass "live .credentials.json unreachable" + CLAUDE_LS=$(p9 "$REAL_SOCKET" ls /claude) + expect_missing "no credentials leaked into /claude" credentials "$CLAUDE_LS" + # The money shot: the mntgen mount every interactive fish sees. + if [ -n "$NS" ]; then + OUT=$(timeout 60 "$NS" --mntgen -- sh -c "ls /mnt/9p/$REAL_NAME && head -c 200 /mnt/9p/$REAL_NAME/claude/history" 2>"$TMP/mntgen.err") + RC=$? + if [ $RC -eq 0 ]; then + expect_contains "mntgen mount lists the agents tree" claude "$OUT" + expect_contains "mntgen mount reads claude/history" claude "$OUT" + else + fail "mntgen money shot (exit $RC)" "$(cat "$TMP/mntgen.err")" + fi + else + echo "skip - mntgen money shot: 9ns not provided" + fi + kill "$DAEMON_PID" 2>/dev/null + wait "$DAEMON_PID" 2>/dev/null + sleep 0.2 + [ -S "$REAL_SOCKET" ] && fail "stop unposts the real name" "socket still there" || pass "stop unposts the real name" + else + fail "daemon posts into the real registry" "$(cat "$TMP/daemon.log")" + fi +fi + +echo "# $PASSED passed, $FAILED failed" +[ "$FAILED" -eq 0 ] diff --git a/9harness/build.zig b/9harness/build.zig deleted file mode 100644 index 4cf84d6..0000000 --- a/9harness/build.zig +++ /dev/null @@ -1,54 +0,0 @@ -//! Build fragment for 9harness: the harness fs daemon (binary `9harness`), -//! a read-only fresh-from-disk 9P2000 view of every harness's state, -//! posted by default under the name `harness`. It is `@import`ed by the -//! root build.zig and called with the root builder, so every `b.path(...)` -//! here is relative to the cloud9 root (hence the `9harness/` prefix), -//! every option is defined by the root and every step it registers lands -//! in the root's step list under the `9harness` prefix. -//! -//! Steps: 9harness, 9harness-test, 9harness-itest. -const std = @import("std"); - -pub const Context = struct { - target: std.Build.ResolvedTarget, - optimize: std.builtin.OptimizeMode, - cloud9: *std.Build.Module, - /// The 9ns binary, for the --mntgen money shot in the integration - /// suite; null when 9ns is disabled, in which case the mntgen - /// section of the suite is skipped. - ns: ?*std.Build.Step.Compile, -}; - -pub const Artifacts = struct { - exe: *std.Build.Step.Compile, - /// `9harness-test`: the tree's unit tests (exclusions, path safety, - /// node ids, vanishing files — all against a fake HOME). - test_step: *std.Build.Step, - /// `9harness-itest`: test/e2e.sh (fixture roots + scratch registry, - /// plus the real-registry end-to-end when the `harness` name is free). - itest_step: *std.Build.Step, -}; - -pub fn add(b: *std.Build, ctx: Context) Artifacts { - const mod = b.createModule(.{ - .root_source_file = b.path("9harness/src/main.zig"), - .target = ctx.target, - .optimize = ctx.optimize, - .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, - }); - const exe = b.addExecutable(.{ .name = "9harness", .root_module = mod }); - const install = b.addInstallArtifact(exe, .{}); - b.getInstallStep().dependOn(&install.step); - b.step("9harness", "Build and install only the harness fs daemon").dependOn(&install.step); - - const test_step = b.step("9harness-test", "Run the 9harness tree's unit tests (fake HOME; never the live roots)"); - test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = mod })).step); - - const itest_step = b.step("9harness-itest", "Run 9harness/test/e2e.sh (posted name, reads, cmp of a transcript, mntgen mount, exclusions, live appends)"); - const run = b.addSystemCommand(&.{"bash"}); - run.addFileArg(b.path("9harness/test/e2e.sh")); - run.addArtifactArg(exe); - if (ctx.ns) |ns| run.addArtifactArg(ns); - itest_step.dependOn(&run.step); - return .{ .exe = exe, .test_step = test_step, .itest_step = itest_step }; -} diff --git a/9harness/docs/DESIGN.md b/9harness/docs/DESIGN.md deleted file mode 100644 index 340f3da..0000000 --- a/9harness/docs/DESIGN.md +++ /dev/null @@ -1,441 +0,0 @@ -# 9harness design - -9harness is the harness fs daemon: a long-running 9P2000 server that -replaces the `zmxify` rc script's introspection half with a unified, -file-shaped, always-fresh view of every AI-agent harness's state on the -machine — Claude Code (`~/.claude`), Codex (`~/.codex`), omp (`~/.omp`), -hermes (`~/.hermes`) and dsh (`~/.dsh`), plus their skills. Everything is -read-only and read live from disk at request time; nothing is cached, so -a session transcript grows as its harness writes it and a new session -appears as soon as its file lands. - -One binary, hosted Linux only (it posts through `cloud9.post`). The -backend is `src/tree.zig`, a `cloud9.fs.Server` backend run by -`cloud9.serve.Runner`; the CLI is `src/main.zig`. Like its siblings -(9proc, 9ns) the build is a fragment in `build.zig` wired into the root -behind `-D9harness` (steps `9harness`, `9harness-test`, `9harness-itest`; -installed by the plain `zig build` next to the other programs). - -## The tree (v1 contract) - -``` -/ read-only root (dr-xr-xr-x) -/pid the daemon's pid, one line -/uptime seconds since start, one line -/claude/ - projects/ mirror of ~/.claude/projects//: one dir - per project, its *.jsonl session transcripts as plain - readable files, memory/ subdirs included - history ~/.claude/history.jsonl (global prompt history) - skills/ mirror of ~/.claude/skills -/codex/ - sessions/ mirror of ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl - session-index ~/.codex/session_index.jsonl - history ~/.codex/history.jsonl -/omp/ mirror of ~/.omp/agent (history.db, agent.db, models.db, - ... served as raw readable blobs; sqlite parsing is a - later step — v1 is files) -/hermes/ mirror of ~/.hermes (state.db raw, logs/) -/dsh/ mirror of ~/.dsh (profiles/, storages/, ... whatever it holds) -/skills/ union view: plain dirs claude/, codex/, omp/ mirroring each - harness's skills dir; a harness without one simply has no entry -``` - -Mirroring is lazy: nothing is walked at startup. A lookup, getattr or -readdir stats and lists the real files at request time. Unknown files and -directories appear as themselves, readable. A file that vanishes between -lookup and read answers ENOENT cleanly; a file that appears is visible on -the next request. - -Writes, creates, removes and setattrs answer EPERM (create/remove/wstat -are not declared as backend features, so the engine refuses them itself; -the write and setattr ops answer `E.PERM`, as does an open for write). -Modes are reported read-only regardless of the real bits: directories -`0o555`, files `0o444`. Sizes and mtimes are real. - -The engine's known fidelity gap applies: directory records carry fixed -modes and length 0, so `9p ls -l` shows placeholders; `9p stat` is the -accurate one. - -## The exclusions (the security boundary) - -The daemon must never become a credential reader for anything that -mounts it. The rule is absolute — `excluded()` in tree.zig, unit-tested — -and applies to every path component of every subtree, by lookup *and* by -readdir, so an excluded name neither resolves nor lists: - -- names containing `credentials`, `token`, `auth` or `secret` - (case-insensitive substrings; "auth" also hides "author-notes" — the - price of never guessing wrong); -- names ending `.key`, `.pem` or `.env`; -- the config/settings files of the mirrored harnesses, which may embed - API keys: `settings.json`, `settings.yaml`, `settings.local.json`, - `config.yml`, `config.yaml`, `config.toml`, `models.yml`, `models.yaml` - (Claude Code's settings.json is not in the tree anyway: /claude serves - only projects/, skills/ and history.jsonl); -- `.ssh`, and `id_rsa`/`id_dsa`/`id_ecdsa`/`id_ed25519` prefixes. - -Symlinks are never served, at any depth: they are an escape hatch around -the root pinning (a symlink out of `~/.claude` would make the daemon read -outside the five roots). A symlink answers ENOENT and is not listed. - -Path resolution is the mechanism that makes that true, and it is not -string composition. `openIn` walks from the pinned base one component at -a time, each opened with `O_NOFOLLOW`: the intermediate components with -`O_PATH|O_DIRECTORY` (so a component swapped for a symlink fails with -ENOTDIR rather than redirecting the walk) and the last with the caller's -flags. Every stat, read and readdir goes through it. Composing -`/` and opening that in one call is what the first version -did, and it was wrong in a way a test could not see: only the final -component's symlink was refused, so a name replaced between the walk and -the read — the plain TOCTOU — served bytes from outside every root, and -a directory above it could redirect the whole path at any time. `O_PATH` -also means a fifo or device left in a harness root is stat'd without its -open ever blocking; a read refuses anything that is not a regular file. - -The base itself comes from $HOME or `--root NAME=PATH` at startup and is -resolved normally (it is configuration, and may legitimately be a link). -Relative paths are composed only from remembered (root, rel) pairs plus -single-segment, engine-vetted names. Client bytes never form a path: -names with `/` or NUL are refused before the filesystem is touched, `.` -and `..` are refused as components inside the walk, and `..` as a lookup -resolves through the backend's own parent map, never the kernel's. - -## Backend shape - -`Harness` (tree.zig) is the backend of `fs.Server(Harness, opts)` on -`serve.Runner` (main.zig): allocation-free request paths, comptime caps, -`Io` passed explicitly, static memory in .bss (the path table and the -per-connection buffers; ~3 MB of tables, untouched pages cost nothing). - -- **Node ids** pack `Node{idx: u8, kind: u8, serial: u48}` in the u64, - zmx-style. Static skeleton nodes (facts, the harness dirs, skills) are - `kind = .top`; every mirrored file is `kind = .path` with - `serial = (slot, generation)` of its table entry. Two walks of the same - file find the same entry, so their node ids — and qid paths — are - equal; an entry freed by its last release takes a new generation, so a - stale forged node id never resolves. -- **The path table** remembers at most 4096 remembered (root, rel) pairs, - each with a Wyhash of the pair for dedupe and a reference count. The - backend declares `features = .{ .references = true }`, so the engine - asks lookups for `.` too and pays every reference back with exactly one - `release`; an entry's refcount reaching zero frees it. Readdir records - name entries without references (a listing does not pin), so a later - walk of the same name finds the same entry and the same node id. A full - table answers `E.NFILE` — on a lookup, and on a listing that cannot - name all its entries (see Readdir). -- **Fresh reads**: a read opens the file, `readPositionalAll`s the window - at the request's offset, and closes it — every read is against the live - file, and the engine handles offsets, so `cat` of a 5 MB transcript - through a mntgen mount works. -- **Readdir** collects a directory's surviving names, sorts them for - cross-read cursor stability, and stages the engine's records - (`node:u64le dir:u8 len:u8 name`), skipping `req.off` records. The - comptime caps (1024 names, 128 KiB of name bytes, the path table) are - **loud**: a directory that would not fit answers `E.NFILE` instead of a - short listing, because a short listing cannot be told apart from a - small directory and is therefore a wrong answer, not a limitation. The - staging buffer filling is different and normal — it ends that read at - the last record that fit, and the next read continues from there; - staging stops at the first record that does not fit rather than packing - a later, shorter one behind it, which would drop that entry from the - listing entirely. A directory that changes between the reads of one - listing may shift its cursor — the fresh-tree trade, accepted in v1. -- **Locking**: one mutex spans `handle` and the reply (the answer's bytes - point into shared staging buffers), so the backend is serialized across - connections. v1 accepts this: it is an introspection fs, not a - throughput service. - -## CLI and process model - -``` -usage: 9harness [--unix PATH | --tcp IP:PORT | --fd N] [--no-post] - [--name NAME] [--root NAME=PATH]... -``` - -- Default: post itself under the name `harness` via - `serve.Runner.listenPosted`, so it appears at - `$XDG_RUNTIME_DIR/9p/harness` and every interactive fish (self-wrapped - in `9ns --mntgen`) sees it at `/mnt/9p/harness` with zero - configuration. `stop()` unposts; the stale-socket protocol covers a - killed daemon. -- `--unix`/`--tcp` listen forms (beside the post, or with `--no-post`), - `--name` to post under another name, `--root NAME=PATH` to pin any of - the five roots elsewhere (defaults are `$HOME/.`; the tests use - fake homes and never the live roots). -- `--fd N` serves exactly one 9P session over the connected stream on - descriptor N (socket activation, `9ns --spawn` handoffs), driving the - engine directly; it posts nothing. -- SIGTERM/SIGINT stop cleanly (unpost); SIGPIPE is ignored. -- No daemonization in v1: run it in a zmx session, the zmx way — - `zmx run harness -d 9harness`. - -## Integration test plan (`test/e2e.sh`) - -Part A runs entirely on fixtures (a fake home pinned by `--root`, a -scratch `XDG_RUNTIME_DIR` registry, plan9port's `9p` as the client): - -1. the posted name is listed in the registry; -2. `9p ls /` shows the roots; a fixture transcript reads back - byte-identical (`cmp` against the source file); the skills union - mirrors the harness tree; -3. credentials-shaped files are unreachable by both listing and reading, - at several depths, in every root; -4. a file appended after the daemon started is visible immediately, and - a new session directory appears in the next listing; -5. the pid fact answers the daemon's pid; writes answer EPERM; -6. the `--unix` and `--fd` listen forms serve; -7. SIGTERM unposts the name. - -Part B is the money shot, read-only against the real roots and the real -`$XDG_RUNTIME_DIR/9p` — only when the `harness` name is free (a live -daemon owning the name skips it, not fails): the posted name in the real -registry, `9p` walks of the live `~/.claude`, a real transcript -byte-identical through the tree, the live `~/.claude/.credentials.json` -unreachable, the `9ns --mntgen` mount of `/mnt/9p/harness` with -`head` of `/claude/history`, and the stop unposting the real name. - -Unit tests (tree.zig) cover the exclusions (including the predicate -itself, spelled out), path-composition safety (slashes, NULs, symlinks, -the `..` chain, NOTDIR), node id packing (same file equal, distinct -files differ, the union identity, id recycling through release), the -vanishing file (ENOENT on lookup, getattr and read), EPERM on -write/setattr/open-for-write, and live visibility — all against a fake -HOME under a test tmp dir, never the real harness roots. - -## Verification - -- `zig build` (installs `9harness` beside 9ns, 9proc-demo, 9web). -- `zig build test` — the library's 80 tests stay green. -- `zig build 9harness-test` — the tree's unit tests. -- `zig build 9harness-itest` — `test/e2e.sh` (fixture + real registry). -- `zig build 9proc-check-freestanding` — the daemon is hosted-only but - must not break the root module. -- `zig build programs-test` / `programs-itest` — the umbrella steps, - which include 9harness's. - -## /active: the derived view (v2) - -v1 mirrors files. `/active` is the first *semantic* layer: one directory -per live agent, normalized across harnesses, so `ls /active` answers -"what is running right now" whatever wrote it. It is the discovery half -of `~/.local/bin/zmxify`, whose resolution ladder it ports. - -``` -/active/ - claude/ - 345104/ - pid 345104 ppid 344980 - started 1790013029 cwd /home/goblin - name goblin-e5 title the session's title - session 1f9a74f7-... model claude-opus-5[1m] - via registry zmx harness (read-write, below) - status busy only where the harness publishes one - transcript the live .jsonl, growing as it writes - agents/ /{model,transcript} - omp/ - 155574/ ... -``` - -A harness directory holds one directory per live process of it, named -by pid. `/proc` spells it that way and so does this: a compound -`claude-345104` would make you parse a name to recover `harness`, which -is already the directory above it, and `/active/omp/*` would not glob. - -There is no `updated` file. The last write to a session is the mtime of -`transcript`, which `stat` already carries; serving it again as its own -file would be the same fact twice, and the copy is the one that goes -stale. - -These are zmxify's picker columns — pid, harness, dir, age, zmx, via, -session, title — as files, which is the point: the script stops -scanning `/proc`, reading fds and querying sqlite itself and just reads -the tree. - -**`status` is a promise only some harnesses make.** It is the harness's -own word for what it is doing, and only Claude Code publishes one -(`busy`, in `sessions/.json`); for every other harness the file is -simply absent, the way a `skills` mount with no target is absent. -Deriving one from the process's `/proc` state would answer a different -question (`S` means "not on a CPU this instant", not "waiting for -you"). For every other harness the honest activity signal is the mtime -of `transcript`. `ls /active/*/*/status` tells you who publishes the -stronger answer. - -An entry is named `-`: unique, stable for the process's -life, and it sorts by harness. The pid is the identity because it is -what `/proc` and every harness's own registry agree on. - -### Finding the agents - -`/proc` is scanned for a process whose `argv[0]` basename is `omp`, -`claude`, `codex`, `hermes` or `dsh`, or a python running -`hermes_cli.main`. A command line carrying one of the daemon words -(`gateway`, `dashboard`, `mcp`, `mcp-server`, `app-server`, -`exec-server`, `serve`, `daemon`, `acp`, `ps`, `render`, `export`, -`__omp_worker_daemon_broker`) is a server or a helper, never a session. -The daemon's own ancestry is excluded, so 9harness can never list or -act on the process tree it lives in. - -Liveness is `/proc/` existing **and** its `stat` field 22 -(starttime) matching the one remembered for the slot. A pid that has -been reused is a different process and does not list; a corpse never -lists. - -### Resolving the session - -Each harness is asked in its own terms — the `fd -> dir -> db` ladder -zmxify worked out, with the route named in `via` so a wrong guess is -visible rather than silent: - -| harness | route | `via` | -|---|---|---| -| claude | `~/.claude/sessions/.json`, which the harness maintains itself: sessionId, cwd, name, status, version, and `procStart` as a pid-reuse guard | `registry` | -| omp | the transcript it holds open under `~/.omp/agent/sessions/`, newest first; else the store directory named after its cwd with `/` becoming `-` | `fd`, `dir` | -| dsh | the `session-/session.jsonl.zstd` it holds open; zstd, so there is no title | `fd` | -| codex | the rollout it holds open — `sessions/YYYY/MM/DD/rollout--.jsonl`, whose *name* carries the session id, so its sqlite is never opened | `fd` | -| hermes | nothing to resolve: see below | `none` | - -A resolved session is checked against the process's cwd before it is -believed (the head of an omp transcript carries `"cwd"`). A process -whose session does not resolve still appears, with everything `/proc` -knows and an empty `session`: the view never pretends to know what it -does not, and a move on it is refused. - -codex's id is the last 36 bytes of the rollout's name, not what -splitting on `-` gives — the timestamp in front of it holds dashes too. - -**hermes is the one harness with no answer here, and it needs none.** -Its sessions live only in `state.db`, with no per-session file to find; -but the only hermes processes that run are the gateway and the -dashboard, and both are daemon-shaped, so neither is a session anything -should move. zmxify resolved a hermes id from sqlite and then declined -to touch the process holding it, for the same reason. If an interactive -hermes ever exists, this is the gap, and it is the one place sqlite -would buy something. - -### Freshness - -A slot remembers only what identifies the agent: harness, pid, -starttime, cwd, session id and transcript path. Every *field* is -re-derived from disk when it is read, so `status` is never stale and a -transcript grows under `cat`. `/proc` is rescanned on a readdir of -`/active` and on a lookup that misses, not on every read. - -### `zmx`: the write path, and why it is not a ctl - -Writing a zmx session name into `/active//zmx` moves that agent into -a zmx session of that name: the zmxify action half, as a file. - -Reading `zmx` gives the `ZMX_SESSION` of the process, empty when it runs -outside zmx. Writing sets it. The file means *which zmx session this -agent lives in*, and writing a name makes that true — state, not a verb -channel. A `ctl` taking words would be the ordinary Plan 9 spelling -(`/proc/n/ctl`), and an executable `zmxify` script served in the tree -would be the zmx `attach` spelling, but a script that shells out to a -local binary is a lie over a remote mount: it would run against a -session that is not on the client's machine. A write is served where -the authority is, so it survives being mounted from anywhere. - -What a write does, in order, refusing before it destroys anything: - -1. The name must be zmx's label charset (`[A-Za-z0-9._-]`) and unused by - a live session, else `EEXIST`. -2. The agent's session must have resolved, else `EPERM`. Nothing is - killed that has nowhere to come back to — zmxify's invariant. -3. `SIGTERM`, then `SIGKILL` after 5s, giving up at 12s with `EIO`. -4. `fork`, `chdir` to the agent's cwd, `exec zmx run -d` with a - **fixed argv per harness** (`claude --resume `, - `omp --resume `, `codex resume `, - `hermes --resume `). No client byte ever reaches `exec`: the - only thing the client supplies is the session name, and it is - validated first. -5. The write returns once `$XDG_RUNTIME_DIR/9p/zmx/` appears - (zmx self-posts), or `EIO` on timeout. - -The agent comes back under a new pid, so the old `/active/` is gone -and a new entry takes its place with `zmx` reading the new name. The -window between the kill and the exec is the same exposure zmxify has -always had, and it is why step 2 comes first. - -**This is the tree's first write path, and it is an escalation**: a -client that can write this file can kill the user's agents and cause a -process to be spawned. It is therefore off unless `--allow-move` is -given, and the read-only daemon stays the default. Everything else in -the tree still answers `EPERM` to writes. - -### Where this is not Plan 9 - -Named, because they are choices rather than oversights: - -* **`via` describes how the server found out**, not something true of - the process. That is diagnostics in the interface. It stays because a - resolution that guesses wrong silently is worse than a wart — it is - why zmxify's picker showed the route too. -* **A write to `zmx` does not change the object, it replaces it.** The - process is killed and another starts under a new pid, so the entry - written to disappears. `/proc/n/ctl` accepting `kill` is at least - honest about being a verb with a consequence. The file still beats an - executable served in the tree, which would lie outright over a remote - mount, so it stays — with this paragraph. -* **A move blocks the whole daemon** for as long as the kill and the - restart take, because one mutex spans `handle` and the reply. A Plan - 9 server keeps answering other fids meanwhile. The engine already has - parked replies and Tflush for exactly this; the move does not use - them yet, and that is the first thing to revisit. -* **`/active` lives inside the mirror** rather than being its own - posted service. "What files exist" and "what is running" are - different jobs, and a stricter reading would separate them; they - share the pinned roots and the parsing glue, so they share a daemon. - -### Replacing zmxify - -There is no replacement script, because the mount is the replacement: - -``` -ls /mnt/9p/harness/active/*/* what is running -cat /mnt/9p/harness/active/claude/345104/session what it is -echo mine > /mnt/9p/harness/active/claude/345104/zmx -zmx attach mine -``` - -`~/.local/bin/zmxify` was 307 lines of rc because there was no -filesystem to ask: it scanned `/proc`, classified command lines, read -fd tables and queried two sqlite databases to work out what the four -lines above now read. All of that moved into the daemon, where it is -tested. Writing a wrapper back on top would only re-derive what the -tree already says, and would become a second, worse interface beside -the real one. - -Two behaviours the old script had are worth knowing, because they are -now the caller's to arrange and both are one line: pick a session name -nothing holds (the registry, `$XDG_RUNTIME_DIR/9p/zmx`, says which are -taken) and `zmx attach` afterwards. - -A third is gone rather than moved. The old script refused to act on -its own ancestry, because it did the killing itself: kill your own -parent and the script dies before it can re-exec, losing the session it -was rescuing. The daemon kills, re-execs and waits for the new session -to post, and completes all of it whether or not the client is still -connected — verified by hanging up immediately after sending the write -and watching the move land anyway. So moving the terminal you are -sitting in works, which is the common case; the shell drops, and you -reattach with the name you wrote. - -### Testability - -`--proc DIR` overrides `/proc` the way `--root NAME=PATH` overrides a -harness root, so the scan, the liveness guard, the resolution ladder and -the exclusions are unit-tested against a fixture process tree and never -against the live machine. The move itself is exercised end-to-end -against a fake harness in `test/e2e.sh`. - -## Out of scope for v1 (documented, not hidden) - -- Write paths (create/write/setattr stay EPERM), resume/attach - automation (the zmxify re-exec half), sqlite parsing (omp/hermes - sessions stay raw blobs), cross-session search, any caching or - invalidation layer. -- Streaming/parked reads (a read answers at once; the transcripts are - plain files), and union-directory semantics beyond the plain - /skills/ dirs. diff --git a/9harness/src/active.zig b/9harness/src/active.zig deleted file mode 100644 index fb55cf6..0000000 --- a/9harness/src/active.zig +++ /dev/null @@ -1,1006 +0,0 @@ -//! The derived view behind `/active`: one record per live agent, -//! normalized across harnesses. -//! -//! v1 of the tree mirrors files. This module answers a different -//! question — *what is running right now* — by reading `/proc` and then -//! asking each harness in its own terms which stored session the -//! process is writing. That ladder (`fd` -> `dir` -> `db`, with the -//! route named so a wrong guess is visible) is `~/.local/bin/zmxify`'s, -//! ported here so the knowledge lives somewhere tested. -//! -//! Everything here composes paths from the proc root, a pinned harness -//! root and a pid. No client byte reaches the filesystem, and the proc -//! root is a parameter (`--proc`) so the scan, the liveness guard and -//! the exclusions are testable against a fixture tree. -const std = @import("std"); -const Io = std.Io; - -// ---- comptime bounds --------------------------------------------------------- - -/// Live agents held at once. A machine running more than this many -/// interactive harnesses at the same time is not the case to optimize. -pub const max_live: usize = 32; -/// Subagents listed under one agent. -pub const max_agents: usize = 32; -/// Longest directory a process may run in. -pub const cwd_capacity: usize = 512; -/// Longest absolute path this module composes (proc root, harness root). -pub const path_capacity: usize = 1024; -/// Longest session id, title, name or model answered. -pub const text_capacity: usize = 160; -/// `-`. -pub const name_capacity: usize = 32; -/// Bytes of a `/proc` file or a transcript head read at once. -pub const probe_capacity: usize = 8192; - -/// The harnesses this view knows how to recognize. -/// The order is `tree.Root`'s, and `tree.zig` asserts that at comptime: -/// the two enums index the same pinned roots, so a mismatch would hand -/// one harness another's base path. -pub const Kind = enum(u8) { - claude, - codex, - omp, - hermes, - dsh, - - pub fn text(k: Kind) []const u8 { - return @tagName(k); - } -}; - -/// How a session was resolved — reported, so a wrong guess is visible -/// rather than silent. `none` means the process is listed with what -/// `/proc` knows and nothing more, and cannot be moved: hermes is the -/// only harness that always lands here. -pub const Via = enum(u8) { - none, - registry, - fd, - dir, - - pub fn text(v: Via) []const u8 { - return @tagName(v); - } -}; - -/// A fixed-capacity byte buffer. Everything this module remembers is -/// bounded at comptime, like the rest of the daemon. -fn Buf(comptime n: usize) type { - return struct { - bytes: [n]u8 = undefined, - len: u16 = 0, - - const Self = @This(); - - pub fn set(b: *Self, v: []const u8) void { - const take = @min(v.len, n); - @memcpy(b.bytes[0..take], v[0..take]); - b.len = @intCast(take); - } - - pub fn slice(b: *const Self) []const u8 { - return b.bytes[0..b.len]; - } - - pub fn clear(b: *Self) void { - b.len = 0; - } - }; -} - -/// One live agent. -pub const Live = struct { - kind: Kind = .claude, - pid: u32 = 0, - ppid: u32 = 0, - /// `/proc//stat` field 22. Two processes can share a pid over - /// time; they cannot share a pid and a start time, so this is what - /// makes a remembered slot safe to trust later. - starttime: u64 = 0, - /// Unix seconds, from the process's start time against boot time. - started: i64 = 0, - via: Via = .none, - name: Buf(name_capacity) = .{}, - cwd: Buf(cwd_capacity) = .{}, - session: Buf(text_capacity) = .{}, - /// Absolute path of the session transcript, empty when unresolved. - transcript: Buf(path_capacity) = .{}, - /// Directory holding the subagent transcripts, empty when the - /// harness has no such layout. - agent_dir: Buf(path_capacity) = .{}, -}; - -// ---- pure helpers: the part that carries zmxify's knowledge ------------------- - -/// A command line word that marks a daemon, a server or a helper — a -/// process wearing the harness's name that is not an interactive -/// session and must never be listed or acted on. -const not_session = [_][]const u8{ - "__omp_worker_daemon_broker", - "acp", - "mcp", - "mcp-server", - "app-server", - "exec-server", - "gateway", - "dashboard", - "serve", - "daemon", - "ps", - "render", - "export", -}; - -/// The last path component of `p`. -pub fn basename(p: []const u8) []const u8 { - if (std.mem.lastIndexOfScalar(u8, p, '/')) |i| return p[i + 1 ..]; - return p; -} - -/// Iterates a `/proc//cmdline`: NUL-separated argv, usually with a -/// trailing NUL. An empty cmdline (a kernel thread) yields nothing. -pub const Argv = struct { - rest: []const u8, - - pub fn init(cmdline: []const u8) Argv { - return .{ .rest = cmdline }; - } - - pub fn next(a: *Argv) ?[]const u8 { - while (a.rest.len > 0 and a.rest[0] == 0) a.rest = a.rest[1..]; - if (a.rest.len == 0) return null; - const end = std.mem.indexOfScalar(u8, a.rest, 0) orelse a.rest.len; - const word = a.rest[0..end]; - a.rest = a.rest[@min(end + 1, a.rest.len)..]; - return word; - } -}; - -/// The harness a command line runs, if any. `argv[0]`'s basename names -/// it, except for hermes, which runs as a python module; any word from -/// `not_session` disqualifies the whole command line. -pub fn harnessOf(cmdline: []const u8) ?Kind { - var it: Argv = .init(cmdline); - const argv0 = it.next() orelse return null; - const exe_name = basename(argv0); - - var found: ?Kind = null; - if (std.mem.eql(u8, exe_name, "claude")) { - found = .claude; - } else if (std.mem.eql(u8, exe_name, "omp")) { - found = .omp; - } else if (std.mem.eql(u8, exe_name, "codex")) { - found = .codex; - } else if (std.mem.eql(u8, exe_name, "hermes")) { - found = .hermes; - } else if (std.mem.eql(u8, exe_name, "dsh")) { - found = .dsh; - } else if (std.mem.startsWith(u8, exe_name, "python") or std.mem.startsWith(u8, exe_name, "pypy")) { - var py: Argv = .init(cmdline); - while (py.next()) |w| { - if (std.mem.endsWith(u8, w, "hermes") or std.mem.endsWith(u8, w, "hermes_cli.main")) { - found = .hermes; - break; - } - } - } - if (found == null) return null; - - var words: Argv = .init(cmdline); - while (words.next()) |w| { - for (not_session) |bad| if (std.mem.eql(u8, w, bad)) return null; - } - return found; -} - -/// Field `n` (1-based) of a `/proc//stat` line. Field 2 is the -/// executable name in parentheses and may itself hold spaces and -/// parentheses, so the fields are counted from the *last* `)`, never by -/// splitting the whole line. Only fields from 3 on are reachable, which -/// is all any caller wants. -pub fn statField(stat: []const u8, n: usize) ?u64 { - if (n < 3) return null; - const close = std.mem.lastIndexOfScalar(u8, stat, ')') orelse return null; - var rest = stat[close + 1 ..]; - var field: usize = 3; // state, the first field after the name - while (true) { - while (rest.len > 0 and rest[0] == ' ') rest = rest[1..]; - if (rest.len == 0) return null; - const end = std.mem.indexOfScalar(u8, rest, ' ') orelse rest.len; - if (field == n) return std.fmt.parseInt(u64, rest[0..end], 10) catch null; - rest = rest[end..]; - field += 1; - } -} - -/// The process's start time in clock ticks since boot. Two processes can -/// share a pid over time; they cannot share a pid and a start time, so -/// this is what makes a remembered slot safe to trust later. -pub fn startTimeOf(stat: []const u8) ?u64 { - return statField(stat, 22); -} - -/// The parent pid. -pub fn parentOf(stat: []const u8) ?u32 { - const v = statField(stat, 4) orelse return null; - return std.math.cast(u32, v); -} - -/// The value of `"key"` in a small flat JSON object: a quoted string -/// without its quotes, or a bare token (a number, `true`, `null`). -/// Nested objects are not searched, which is what the callers want — -/// these are the harnesses' own one-level session records. -pub fn jsonField(text: []const u8, key: []const u8) ?[]const u8 { - var quoted_buf: [64]u8 = undefined; - if (key.len + 2 > quoted_buf.len) return null; - quoted_buf[0] = '"'; - @memcpy(quoted_buf[1 .. 1 + key.len], key); - quoted_buf[1 + key.len] = '"'; - const quoted = quoted_buf[0 .. key.len + 2]; - - var from: usize = 0; - while (std.mem.indexOfPos(u8, text, from, quoted)) |at| { - from = at + quoted.len; - var rest = text[from..]; - while (rest.len > 0 and (rest[0] == ' ' or rest[0] == '\t')) rest = rest[1..]; - if (rest.len == 0 or rest[0] != ':') continue; - rest = rest[1..]; - while (rest.len > 0 and (rest[0] == ' ' or rest[0] == '\t')) rest = rest[1..]; - if (rest.len == 0) return null; - if (rest[0] == '"') { - const end = std.mem.indexOfScalar(u8, rest[1..], '"') orelse return null; - return rest[1 .. 1 + end]; - } - if (rest[0] == '{' or rest[0] == '[') continue; // not a leaf; keep looking - const end = std.mem.indexOfAny(u8, rest, ",}\n\r \t") orelse rest.len; - if (end == 0) return null; - return rest[0..end]; - } - return null; -} - -/// The session id in a transcript's file name. omp writes -/// `_.jsonl`, so the id is what follows the last `_`. -/// codex writes `rollout--.jsonl`, where the -/// timestamp holds dashes too, so the id is the last 36 bytes rather -/// than anything found by splitting. Claude Code writes -/// `.jsonl`, all of it the id. -pub fn sessionIdOf(kind: Kind, file_name: []const u8) []const u8 { - var stem = file_name; - if (std.mem.endsWith(u8, stem, ".jsonl")) stem = stem[0 .. stem.len - ".jsonl".len]; - switch (kind) { - .omp => if (std.mem.lastIndexOfScalar(u8, stem, '_')) |i| return stem[i + 1 ..], - .codex => { - const uuid_len = "01a0bcd2-5653-7cc0-aa36-676f6467a72a".len; - if (stem.len >= uuid_len) return stem[stem.len - uuid_len ..]; - }, - else => {}, - } - return stem; -} - -/// The store directory a harness derives from a working directory. -/// omp strips `$HOME` and turns every `/` into `-`; Claude Code turns -/// every character that is not a letter or a digit into `-`, over the -/// whole path. Returns the slug written into `out`. -pub fn slugOf(kind: Kind, cwd: []const u8, home: []const u8, out: []u8) ?[]const u8 { - var path = cwd; - if (kind == .omp and home.len > 0 and std.mem.startsWith(u8, path, home)) { - path = path[home.len..]; - } - if (path.len > out.len) return null; - for (path, 0..) |c, i| { - out[i] = switch (kind) { - .omp => if (c == '/') '-' else c, - else => if (std.ascii.isAlphanumeric(c)) c else '-', - }; - } - return out[0..path.len]; -} - -/// Is `name` a zmx session name? zmx allows only these bytes in a -/// label, and the move path never composes anything else. -pub fn legalZmxName(name: []const u8) bool { - if (name.len == 0 or name.len > 64) return false; - for (name) |c| { - if (!std.ascii.isAlphanumeric(c) and c != '.' and c != '_' and c != '-') return false; - } - return true; -} - - -// ---- reading /proc and the harness stores ------------------------------------ - -/// Where the view reads from. Every path below is composed from these -/// and a pid; nothing here is ever built from a client's bytes. -pub const Sources = struct { - io: Io, - /// The proc filesystem root — `/proc`, or a fixture tree under test. - proc: []const u8, - /// $HOME, for the store-slug spellings. - home: []const u8, - /// The pinned harness roots, indexed by `@intFromEnum(Kind)`; an - /// empty base means that harness is unreachable. - roots: [5][]const u8 = @splat(""), - /// The daemon's own pid: it and its ancestors are never listed, so - /// 9harness can neither show nor act on the tree it lives in. Zero - /// disables the check (fixtures have no ancestry). - self_pid: u32 = 0, - /// Seconds between the epoch and boot, from `/proc/stat`'s `btime`. - /// Zero when it could not be read; `started` is then zero too. - btime: i64 = 0, -}; - -/// USER_HZ: the unit of `/proc//stat`'s time fields. Linux reports -/// them in hundredths of a second whatever CONFIG_HZ is. -const user_hz: u64 = 100; - -fn readSmall(io: Io, path: []const u8, buf: []u8) ?[]const u8 { - const out = Io.Dir.readFile(.cwd(), io, path, buf) catch return null; - return out; -} - -fn linkOf(io: Io, path: []const u8, buf: []u8) ?[]const u8 { - const n = Io.Dir.readLinkAbsolute(io, path, buf) catch return null; - return buf[0..n]; -} - -fn statOf(io: Io, path: []const u8) ?Io.Dir.Stat { - return Io.Dir.statFile(.cwd(), io, path, .{ .follow_symlinks = true }) catch null; -} - -fn mtimeOf(io: Io, path: []const u8) i64 { - const st = statOf(io, path) orelse return 0; - return st.mtime.toSeconds(); -} - -/// `parts` joined with `/` into `buf`, or null when they do not fit. -fn join(buf: []u8, parts: []const []const u8) ?[]const u8 { - var n: usize = 0; - for (parts, 0..) |p, i| { - if (i > 0) { - if (n + 1 > buf.len) return null; - buf[n] = '/'; - n += 1; - } - if (n + p.len > buf.len) return null; - @memcpy(buf[n..][0..p.len], p); - n += p.len; - } - return buf[0..n]; -} - -/// `` of a harness, or null when it is not pinned. -fn rootOf(s: Sources, k: Kind) ?[]const u8 { - const base = s.roots[@intFromEnum(k)]; - return if (base.len == 0) null else base; -} - -/// The directory a harness keeps its session transcripts under. -fn sessionRoot(s: Sources, k: Kind, buf: []u8) ?[]const u8 { - const base = rootOf(s, k) orelse return null; - return switch (k) { - // The omp root is ~/.omp and its mirror hangs off agent/. - .omp => join(buf, &.{ base, "agent", "sessions" }), - .claude => join(buf, &.{ base, "projects" }), - .dsh => join(buf, &.{ base, "sessions" }), - // codex names each rollout after its session, so the file it - // holds open is the whole answer — no sqlite needed. - .codex => join(buf, &.{ base, "sessions" }), - // hermes keeps its sessions only in sqlite, and the only hermes - // processes that run are the gateway and the dashboard, which - // are never sessions. Nothing to resolve. - .hermes => null, - }; -} - -// ---- the table ---------------------------------------------------------------- - -pub const Slot = struct { - used: bool = false, - /// Bumped whenever the slot comes to hold a different process, so a - /// node id minted for the old one stops resolving. - gen: u8 = 0, - seen: bool = false, - live: Live = .{}, -}; - -pub const Table = struct { - slots: [max_live]Slot = @splat(.{}), - - /// The slot holding this (pid, starttime), if any. - fn slotOfPid(t: *Table, pid: u32, starttime: u64) ?*Slot { - for (&t.slots) |*sl| { - if (sl.used and sl.live.pid == pid and sl.live.starttime == starttime) return sl; - } - return null; - } - - fn freeSlot(t: *Table) ?*Slot { - for (&t.slots) |*sl| if (!sl.used) return sl; - return null; - } - - /// The live record at `index`, when its generation still matches. - pub fn at(t: *Table, index: usize, gen: u8) ?*Live { - if (index >= t.slots.len) return null; - const sl = &t.slots[index]; - if (!sl.used or sl.gen != gen) return null; - return &sl.live; - } - - pub fn indexOf(t: *Table, l: *const Live) usize { - const base = @intFromPtr(&t.slots[0]); - return (@intFromPtr(l) - @sizeOf(Slot) + @sizeOf(Slot) - @offsetOf(Slot, "live") - base) / @sizeOf(Slot); - } - - /// Rescans `/proc`. Slots keep their index and generation while the - /// same process is still there, so a handle opened before the scan - /// keeps working; a slot that comes to hold a different process - /// bumps its generation and the old node ids stop resolving. - pub fn scan(t: *Table, s: Sources) void { - for (&t.slots) |*sl| sl.seen = false; - - var anc: [64]u32 = @splat(0); - const anc_n = ancestry(s, &anc); - - var buf: [path_capacity]u8 = undefined; - if (Io.Dir.openDirAbsolute(s.io, s.proc, .{ .iterate = true })) |dir| { - defer Io.Dir.close(dir, s.io); - var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; - var it = Io.Dir.Reader.init(dir, &rb); - while (true) { - const entry = (it.next(s.io) catch break) orelse break; - const pid = std.fmt.parseInt(u32, entry.name, 10) catch continue; - var skip = false; - for (anc[0..anc_n]) |a| if (a == pid) { - skip = true; - }; - if (skip) continue; - t.consider(s, pid, &buf); - } - } else |_| {} - - for (&t.slots) |*sl| { - if (sl.used and !sl.seen) { - sl.used = false; - sl.gen +%= 1; - } - } - } - - /// Looks at one pid and takes it if it is an agent. - fn consider(t: *Table, s: Sources, pid: u32, buf: []u8) void { - var num: [24]u8 = undefined; - const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch return; - var dir_buf: [path_capacity]u8 = undefined; - const pid_dir = join(&dir_buf, &.{ s.proc, pid_text }) orelse return; - - var stat_buf: [probe_capacity]u8 = undefined; - const stat_path = join(buf, &.{ pid_dir, "stat" }) orelse return; - const stat = readSmall(s.io, stat_path, &stat_buf) orelse return; - const starttime = startTimeOf(stat) orelse return; - - var cmd_buf: [probe_capacity]u8 = undefined; - const cmd_path = join(buf, &.{ pid_dir, "cmdline" }) orelse return; - const cmdline = readSmall(s.io, cmd_path, &cmd_buf) orelse return; - const kind = harnessOf(cmdline) orelse return; - - // Already known and still the same process: keep the slot and - // everything resolved for it. - if (t.slotOfPid(pid, starttime)) |sl| { - sl.seen = true; - return; - } - - const sl = t.freeSlot() orelse return; // full: the rest simply do not list - sl.* = .{ .used = true, .gen = sl.gen +% 1, .seen = true, .live = .{} }; - const l = &sl.live; - l.kind = kind; - l.pid = pid; - l.ppid = parentOf(stat) orelse 0; - l.starttime = starttime; - l.started = if (s.btime == 0) 0 else s.btime + @as(i64, @intCast(starttime / user_hz)); - - var name_buf: [name_capacity]u8 = undefined; - l.name.set(std.fmt.bufPrint(&name_buf, "{s}-{d}", .{ kind.text(), pid }) catch kind.text()); - - const cwd_path = join(buf, &.{ pid_dir, "cwd" }) orelse return; - var cwd_buf: [cwd_capacity]u8 = undefined; - if (linkOf(s.io, cwd_path, &cwd_buf)) |cwd| l.cwd.set(cwd); - - resolveSession(l, s, pid_dir); - } -}; - -/// The daemon's own pid and every parent of it, so the tree it lives in -/// is never listed. Returns how many were written. -fn ancestry(s: Sources, out: []u32) usize { - if (s.self_pid == 0) return 0; - var n: usize = 0; - var pid = s.self_pid; - var buf: [path_capacity]u8 = undefined; - var stat_buf: [probe_capacity]u8 = undefined; - while (pid > 1 and n < out.len) { - out[n] = pid; - n += 1; - var num: [24]u8 = undefined; - const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch break; - const path = join(&buf, &.{ s.proc, pid_text, "stat" }) orelse break; - const stat = readSmall(s.io, path, &stat_buf) orelse break; - const parent = parentOf(stat) orelse break; - if (parent == 0 or parent == pid) break; - pid = parent; - } - return n; -} - -/// Asks the harness, in its own terms, which stored session this -/// process is writing. Leaves `via = .none` when it cannot tell: the -/// view never invents a session it did not find. -fn resolveSession(l: *Live, s: Sources, pid_dir: []const u8) void { - switch (l.kind) { - .claude => resolveClaude(l, s), - .omp => resolveOmp(l, s, pid_dir), - .dsh, .codex => resolveByFd(l, s, pid_dir), - // hermes has no per-session file to find. - .hermes => {}, - } -} - -/// Claude Code maintains `sessions/.json` itself, so there is -/// nothing to guess — only to check. `procStart` must match the -/// process's start time, or the record belongs to a dead predecessor -/// that happened to share the pid. -fn resolveClaude(l: *Live, s: Sources) void { - const base = rootOf(s, .claude) orelse return; - var buf: [path_capacity]u8 = undefined; - var num: [24]u8 = undefined; - const file = std.fmt.bufPrint(&num, "{d}.json", .{l.pid}) catch return; - const path = join(&buf, &.{ base, "sessions", file }) orelse return; - var text_buf: [probe_capacity]u8 = undefined; - const text = readSmall(s.io, path, &text_buf) orelse return; - - const proc_start = jsonField(text, "procStart") orelse return; - const claimed = std.fmt.parseInt(u64, proc_start, 10) catch return; - if (claimed != l.starttime) return; // a record left by a reused pid - - const sid = jsonField(text, "sessionId") orelse return; - if (sid.len == 0 or sid.len > text_capacity) return; - l.session.set(sid); - l.via = .registry; - - var slug_buf: [cwd_capacity]u8 = undefined; - const slug = slugOf(.claude, l.cwd.slice(), s.home, &slug_buf) orelse return; - var tr_buf: [path_capacity]u8 = undefined; - var file_buf: [text_capacity + 8]u8 = undefined; - const tr_name = std.fmt.bufPrint(&file_buf, "{s}.jsonl", .{sid}) catch return; - const tr = join(&tr_buf, &.{ base, "projects", slug, tr_name }) orelse return; - if (statOf(s.io, tr) != null) l.transcript.set(tr); -} - -/// omp: the transcript it holds open, else the store directory named -/// after its working directory. -fn resolveOmp(l: *Live, s: Sources, pid_dir: []const u8) void { - resolveByFd(l, s, pid_dir); - if (l.via == .none) resolveOmpByDir(l, s); - if (l.transcript.len == 0) return; - const tr = l.transcript.slice(); - l.session.set(sessionIdOf(.omp, basename(tr))); - // The subagents are sibling files in the directory named after the - // session file, one `.jsonl` each. - if (std.mem.endsWith(u8, tr, ".jsonl")) { - l.agent_dir.set(tr[0 .. tr.len - ".jsonl".len]); - } -} - -fn resolveOmpByDir(l: *Live, s: Sources) void { - var root_buf: [path_capacity]u8 = undefined; - const root = sessionRoot(s, .omp, &root_buf) orelse return; - var slug_buf: [cwd_capacity]u8 = undefined; - const slug = slugOf(.omp, l.cwd.slice(), s.home, &slug_buf) orelse return; - var dir_buf: [path_capacity]u8 = undefined; - const dir_path = join(&dir_buf, &.{ root, slug }) orelse return; - var newest_buf: [path_capacity]u8 = undefined; - const newest = newestUnder(s.io, dir_path, ".jsonl", &newest_buf) orelse return; - l.transcript.set(newest); - l.via = .dir; -} - -/// The newest entry of `dir_path` whose name ends in `suffix`, as a -/// full path written into `out`. -fn newestUnder(io: Io, dir_path: []const u8, suffix: []const u8, out: []u8) ?[]const u8 { - const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch return null; - defer Io.Dir.close(dir, io); - var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; - var it = Io.Dir.Reader.init(dir, &rb); - var best: i64 = -1; - var best_len: usize = 0; - var probe: [path_capacity]u8 = undefined; - while (true) { - const entry = (it.next(io) catch break) orelse break; - if (!std.mem.endsWith(u8, entry.name, suffix)) continue; - const path = join(&probe, &.{ dir_path, entry.name }) orelse continue; - const when = mtimeOf(io, path); - if (when <= best) continue; - if (path.len > out.len) continue; - @memcpy(out[0..path.len], path); - best = when; - best_len = path.len; - } - return if (best_len == 0) null else out[0..best_len]; -} - -/// The route that needs no knowledge of how a harness names its store: -/// whatever session file the process is holding open. The newest wins, -/// so a harness that keeps an old transcript open for reading does not -/// outvote the one it is writing. -fn resolveByFd(l: *Live, s: Sources, pid_dir: []const u8) void { - var root_buf: [path_capacity]u8 = undefined; - const root = sessionRoot(s, l.kind, &root_buf) orelse return; - - var fd_buf: [path_capacity]u8 = undefined; - const fd_dir_path = join(&fd_buf, &.{ pid_dir, "fd" }) orelse return; - const fd_dir = Io.Dir.openDirAbsolute(s.io, fd_dir_path, .{ .iterate = true }) catch return; - defer Io.Dir.close(fd_dir, s.io); - var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; - var it = Io.Dir.Reader.init(fd_dir, &rb); - - var best: i64 = -1; - var best_path: [path_capacity]u8 = undefined; - var best_len: usize = 0; - while (true) { - const entry = (it.next(s.io) catch break) orelse break; - var link_path_buf: [path_capacity]u8 = undefined; - const link_path = join(&link_path_buf, &.{ fd_dir_path, entry.name }) orelse continue; - var target_buf: [path_capacity]u8 = undefined; - const target = linkOf(s.io, link_path, &target_buf) orelse continue; - if (!std.mem.startsWith(u8, target, root)) continue; - if (target.len <= root.len or target[root.len] != '/') continue; - if (!sessionFile(l.kind, target[root.len + 1 ..])) continue; - const when = mtimeOf(s.io, target); - if (when <= best or target.len > best_path.len) continue; - @memcpy(best_path[0..target.len], target); - best = when; - best_len = target.len; - } - if (best_len == 0) return; - const found = best_path[0..best_len]; - l.transcript.set(found); - l.via = .fd; - if (l.kind == .codex) l.session.set(sessionIdOf(.codex, basename(found))); - if (l.kind == .dsh) { - // ~/.dsh/sessions//session-/session.jsonl.zstd: - // the id is the directory the transcript sits in. - const rel = found[root.len + 1 ..]; - var parts = std.mem.splitScalar(u8, rel, '/'); - _ = parts.next(); - if (parts.next()) |id| l.session.set(id); - } -} - -/// Is this path, relative to the harness's session root, one of its -/// session transcripts rather than a subagent's or something else? -fn sessionFile(kind: Kind, rel: []const u8) bool { - var depth: usize = 0; - var parts = std.mem.splitScalar(u8, rel, '/'); - while (parts.next()) |_| depth += 1; - return switch (kind) { - // /.jsonl exactly: a third component is a subagent. - .omp => depth == 2 and std.mem.endsWith(u8, rel, ".jsonl"), - .claude => depth == 2 and std.mem.endsWith(u8, rel, ".jsonl"), - .dsh => depth == 3 and std.mem.startsWith(u8, rel, "-"), - // YYYY/MM/DD/rollout--.jsonl - .codex => depth == 4 and std.mem.endsWith(u8, rel, ".jsonl") and - std.mem.startsWith(u8, basename(rel), "rollout-"), - .hermes => false, - }; -} - - -// ---- fields derived when they are read --------------------------------------- - -/// The value of `key` in a NUL-separated environment block, as -/// `/proc//environ` stores it. -pub fn envValue(environ: []const u8, key: []const u8) ?[]const u8 { - var it: Argv = .init(environ); - while (it.next()) |pair| { - if (pair.len <= key.len) continue; - if (!std.mem.startsWith(u8, pair, key)) continue; - if (pair[key.len] != '=') continue; - return pair[key.len + 1 ..]; - } - return null; -} - -/// The title key each harness writes into the head of a transcript. -fn titleKey(k: Kind) []const u8 { - return switch (k) { - .claude => "aiTitle", - else => "title", - }; -} - -/// A field from the first few records of a transcript, where the -/// harnesses put the session's title and the model it runs. Only the -/// head is read: these records are written when the session opens. -pub fn headField(io: Io, kind: Kind, transcript: []const u8, which: enum { title, model }, out: []u8) ?[]const u8 { - if (transcript.len == 0) return null; - var head: [probe_capacity]u8 = undefined; - const text = readSmall(io, transcript, &head) orelse return null; - const key = switch (which) { - .title => titleKey(kind), - .model => "model", - }; - const v = jsonField(text, key) orelse return null; - if (v.len == 0 or v.len > out.len) return null; - @memcpy(out[0..v.len], v); - return out[0..v.len]; -} - -/// The zmx session a process runs inside, from its environment. -pub fn zmxOf(s: Sources, pid: u32, out: []u8) ?[]const u8 { - var num: [24]u8 = undefined; - const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch return null; - var path_buf: [path_capacity]u8 = undefined; - const path = join(&path_buf, &.{ s.proc, pid_text, "environ" }) orelse return null; - var env_buf: [probe_capacity]u8 = undefined; - const env = readSmall(s.io, path, &env_buf) orelse return null; - const v = envValue(env, "ZMX_SESSION") orelse return null; - if (v.len == 0 or v.len > out.len) return null; - @memcpy(out[0..v.len], v); - return out[0..v.len]; -} - -/// A field of Claude Code's own session record, which is the only -/// harness that publishes a name and a status. -pub fn registryField(s: Sources, l: *const Live, key: []const u8, out: []u8) ?[]const u8 { - if (l.kind != .claude) return null; - const base = rootOf(s, .claude) orelse return null; - var num: [24]u8 = undefined; - const file = std.fmt.bufPrint(&num, "{d}.json", .{l.pid}) catch return null; - var path_buf: [path_capacity]u8 = undefined; - const path = join(&path_buf, &.{ base, "sessions", file }) orelse return null; - var text_buf: [probe_capacity]u8 = undefined; - const text = readSmall(s.io, path, &text_buf) orelse return null; - // The record must still describe this process, not a reused pid. - const proc_start = jsonField(text, "procStart") orelse return null; - const claimed = std.fmt.parseInt(u64, proc_start, 10) catch return null; - if (claimed != l.starttime) return null; - const v = jsonField(text, key) orelse return null; - if (v.len == 0 or v.len > out.len) return null; - @memcpy(out[0..v.len], v); - return out[0..v.len]; -} - -/// Is this process still the one the slot remembers? A pid alone is not -/// an identity: the kernel reuses them. -pub fn stillAlive(s: Sources, l: *const Live) bool { - var num: [24]u8 = undefined; - const pid_text = std.fmt.bufPrint(&num, "{d}", .{l.pid}) catch return false; - var path_buf: [path_capacity]u8 = undefined; - const path = join(&path_buf, &.{ s.proc, pid_text, "stat" }) orelse return false; - var stat_buf: [probe_capacity]u8 = undefined; - const stat = readSmall(s.io, path, &stat_buf) orelse return false; - const now = startTimeOf(stat) orelse return false; - return now == l.starttime; -} - -/// Seconds between the epoch and boot, from `/proc/stat`'s `btime` -/// line. Zero when it cannot be read. -pub fn bootTime(io: Io, proc: []const u8) i64 { - var path_buf: [path_capacity]u8 = undefined; - const path = join(&path_buf, &.{ proc, "stat" }) orelse return 0; - var buf: [probe_capacity]u8 = undefined; - const text = readSmall(io, path, &buf) orelse return 0; - var lines = std.mem.splitScalar(u8, text, '\n'); - while (lines.next()) |line| { - if (!std.mem.startsWith(u8, line, "btime ")) continue; - return std.fmt.parseInt(i64, std.mem.trim(u8, line["btime ".len..], " \r"), 10) catch 0; - } - return 0; -} - - -// ---- subagents ---------------------------------------------------------------- - -/// Longest subagent name answered. -pub const agent_name_capacity: usize = 64; - -/// The subagents of one session: omp writes each as `.jsonl` in a -/// directory named after the session file. Sorted, so a listing that -/// spans several reads keeps its cursor. -pub const Agents = struct { - names: [max_agents][agent_name_capacity]u8 = undefined, - lens: [max_agents]u8 = @splat(0), - count: usize = 0, - - pub fn name(a: *const Agents, i: usize) []const u8 { - if (i >= a.count) return ""; - return a.names[i][0..a.lens[i]]; - } - - pub fn indexOf(a: *const Agents, want: []const u8) ?usize { - for (0..a.count) |i| if (std.mem.eql(u8, a.name(i), want)) return i; - return null; - } -}; - -pub fn agentsOf(io: Io, l: *const Live, out: *Agents) void { - out.count = 0; - const dir_path = l.agent_dir.slice(); - if (dir_path.len == 0) return; - const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch return; - defer Io.Dir.close(dir, io); - var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; - var it = Io.Dir.Reader.init(dir, &rb); - while (out.count < max_agents) { - const entry = (it.next(io) catch break) orelse break; - if (!std.mem.endsWith(u8, entry.name, ".jsonl")) continue; - const stem = entry.name[0 .. entry.name.len - ".jsonl".len]; - if (stem.len == 0 or stem.len > agent_name_capacity) continue; - @memcpy(out.names[out.count][0..stem.len], stem); - out.lens[out.count] = @intCast(stem.len); - out.count += 1; - } - // Insertion sort: the list is at most `max_agents` long and already - // nearly ordered, and this keeps the struct allocation-free. - var i: usize = 1; - while (i < out.count) : (i += 1) { - var j = i; - while (j > 0 and std.mem.order(u8, out.name(j), out.name(j - 1)) == .lt) : (j -= 1) { - std.mem.swap([agent_name_capacity]u8, &out.names[j], &out.names[j - 1]); - std.mem.swap(u8, &out.lens[j], &out.lens[j - 1]); - } - } -} - -/// The transcript path of subagent `i`, written into `out`. -pub fn agentPath(l: *const Live, a: *const Agents, i: usize, out: []u8) ?[]const u8 { - if (i >= a.count) return null; - var name_buf: [agent_name_capacity + 8]u8 = undefined; - const file = std.fmt.bufPrint(&name_buf, "{s}.jsonl", .{a.name(i)}) catch return null; - return join(out, &.{ l.agent_dir.slice(), file }); -} - -// ---- unit tests --------------------------------------------------------------- - -const testing = std.testing; - -/// A `/proc//cmdline`: NUL-separated argv, written out literally so -/// the tests read the way the kernel stores it. -fn cmd(comptime words: []const []const u8) []const u8 { - comptime var out: []const u8 = ""; - inline for (words) |w| out = out ++ w ++ "\x00"; - return out; -} - -test "active: a command line names its harness, or nothing" { - try testing.expectEqual(Kind.claude, harnessOf(cmd(&.{ "claude", "--dangerously-skip-permissions" })).?); - try testing.expectEqual(Kind.omp, harnessOf(cmd(&.{"omp"})).?); - try testing.expectEqual(Kind.codex, harnessOf(cmd(&.{ "/usr/bin/codex", "resume" })).?); - try testing.expectEqual(Kind.dsh, harnessOf(cmd(&.{"/home/goblin/.local/bin/dsh"})).?); - // hermes runs as a python module, named by a later word. - try testing.expectEqual(Kind.hermes, harnessOf(cmd(&.{ "/venv/bin/python", "-m", "hermes_cli.main" })).?); - // Not a harness at all. - try testing.expect(harnessOf(cmd(&.{ "bash", "-c", "claude" })) == null); - try testing.expect(harnessOf("") == null); - try testing.expect(harnessOf("\x00\x00") == null); -} - -test "active: a daemon or helper wearing a harness name is never a session" { - // The live machine's own gateway and dashboard, which must not list. - try testing.expect(harnessOf(cmd(&.{ "python3", "-m", "hermes_cli.main", "gateway", "run" })) == null); - try testing.expect(harnessOf(cmd(&.{ "python3", "/x/hermes-dashboard", "dashboard" })) == null); - try testing.expect(harnessOf(cmd(&.{ "claude", "mcp-server" })) == null); - try testing.expect(harnessOf(cmd(&.{ "omp", "__omp_worker_daemon_broker" })) == null); - try testing.expect(harnessOf(cmd(&.{ "codex", "app-server" })) == null); - // A directory that merely contains the word is still a session: the - // exclusion matches a whole argument, never a substring. - try testing.expectEqual(Kind.claude, harnessOf(cmd(&.{ "claude", "--cwd", "/home/goblin/serve-me" })).?); -} - -test "active: starttime is counted from the last ')' in a stat line" { - // Fields: pid (comm) state ppid ... starttime is field 22. - var line: [512]u8 = undefined; - var w = Io.Writer.fixed(&line); - try w.print("345104 (claude) S 344980", .{}); - for (5..22) |i| try w.print(" {d}", .{i * 10}); - try w.print(" 1625063 rest of the line", .{}); - try testing.expectEqual(@as(u64, 1625063), startTimeOf(w.buffered()).?); - - // A process whose name holds spaces and parentheses — the reason - // the fields are never counted from the start of the line. - var nasty: [512]u8 = undefined; - var nw = Io.Writer.fixed(&nasty); - try nw.print("42 (we i)rd (name) ) R 1", .{}); - for (5..22) |i| try nw.print(" {d}", .{i}); - try nw.print(" 999 x", .{}); - try testing.expectEqual(@as(u64, 999), startTimeOf(nw.buffered()).?); - - try testing.expect(startTimeOf("no parens here") == null); - try testing.expect(startTimeOf("1 (short) S 2 3") == null); -} - -test "active: json fields come out of a harness's own session record" { - // The shape Claude Code writes to ~/.claude/sessions/.json. - const rec = - \\{ - \\ "pid": 345104, - \\ "sessionId": "1f9a74f7-b04f-4092-8b98-df11316e03c0", - \\ "cwd": "/home/goblin", - \\ "version": "2.1.278", - \\ "peerFeatures": ["notify_idle", "artifact_yield"], - \\ "name": "goblin-e5", - \\ "status": "busy", - \\ "procStart": "1625063" - \\} - ; - try testing.expectEqualStrings("345104", jsonField(rec, "pid").?); - try testing.expectEqualStrings("1f9a74f7-b04f-4092-8b98-df11316e03c0", jsonField(rec, "sessionId").?); - try testing.expectEqualStrings("/home/goblin", jsonField(rec, "cwd").?); - try testing.expectEqualStrings("goblin-e5", jsonField(rec, "name").?); - try testing.expectEqualStrings("busy", jsonField(rec, "status").?); - try testing.expectEqualStrings("1625063", jsonField(rec, "procStart").?); - // An array or object value is skipped, not half-parsed. - try testing.expect(jsonField(rec, "peerFeatures") == null); - try testing.expect(jsonField(rec, "absent") == null); - // A key that is a prefix of another must not match it. - try testing.expect(jsonField("{\"sessionIdent\":\"x\"}", "sessionId") == null); -} - -test "active: session ids and store slugs follow each harness's spelling" { - try testing.expectEqualStrings( - "01a0c46a-7a06-7676-aef1-add9d9340c83", - sessionIdOf(.omp, "2026-09-21T14-41-47-526Z_01a0c46a-7a06-7676-aef1-add9d9340c83.jsonl"), - ); - try testing.expectEqualStrings( - "42898558-2fa0-41f2-a2c6-1357e5cf6836", - sessionIdOf(.claude, "42898558-2fa0-41f2-a2c6-1357e5cf6836.jsonl"), - ); - // codex's timestamp holds dashes too, so the id is the last 36 - // bytes and never what splitting on `-` would give. - try testing.expectEqualStrings( - "01a0bcd2-5653-7cc0-aa36-676f6467a72a", - sessionIdOf(.codex, "rollout-2026-09-20T00-18-16-01a0bcd2-5653-7cc0-aa36-676f6467a72a.jsonl"), - ); - - var buf: [256]u8 = undefined; - // omp strips $HOME and keeps everything but the slashes. - try testing.expectEqualStrings( - "-00-projects-0x4200.cafe-cloud9", - slugOf(.omp, "/home/goblin/00-projects/0x4200.cafe/cloud9", "/home/goblin", &buf).?, - ); - // Claude Code dashes every byte that is not a letter or a digit. - try testing.expectEqualStrings( - "-home-goblin-00-projects-0x4200-cafe-cloud9", - slugOf(.claude, "/home/goblin/00-projects/0x4200.cafe/cloud9", "/home/goblin", &buf).?, - ); - // A path longer than the buffer is refused, never truncated into a - // slug that would name the wrong store. - var tiny: [4]u8 = undefined; - try testing.expect(slugOf(.omp, "/home/goblin/somewhere", "/home/goblin", &tiny) == null); -} - -test "active: a zmx session name is checked before it is ever used" { - try testing.expect(legalZmxName("harness")); - try testing.expect(legalZmxName("claude-cloud9.2_x")); - try testing.expect(!legalZmxName("")); - try testing.expect(!legalZmxName("has space")); - try testing.expect(!legalZmxName("../escape")); - try testing.expect(!legalZmxName("semi;colon")); - try testing.expect(!legalZmxName("nul\x00byte")); - try testing.expect(!legalZmxName("x" ** 65)); -} - -test "active: an environment block answers by key, not by prefix" { - const env = "PATH=/usr/bin\x00ZMX_SESSION=harness\x00HOME=/home/goblin\x00"; - try testing.expectEqualStrings("harness", envValue(env, "ZMX_SESSION").?); - try testing.expectEqualStrings("/home/goblin", envValue(env, "HOME").?); - try testing.expect(envValue(env, "ZMX") == null); // a prefix is not the key - try testing.expect(envValue(env, "NOPE") == null); - try testing.expect(envValue("", "HOME") == null); - // An entry with no value is a value of zero length, not a miss. - try testing.expectEqualStrings("", envValue("EMPTY=\x00", "EMPTY").?); -} diff --git a/9harness/src/main.zig b/9harness/src/main.zig deleted file mode 100644 index fe0a07b..0000000 --- a/9harness/src/main.zig +++ /dev/null @@ -1,379 +0,0 @@ -//! 9harness: the harness fs daemon — a long-running 9P2000 server that -//! mirrors every AI-agent harness's state (Claude Code, Codex, omp, -//! hermes, dsh + their skills) as one read-only, fresh-from-disk tree. -//! See docs/DESIGN.md for the tree contract and src/tree.zig for the tree. -//! -//! By default it posts itself under the name `harness` with -//! `serve.Runner.listenPosted`, so it appears at $XDG_RUNTIME_DIR/9p/harness -//! and every interactive fish (self-wrapped in `9ns --mntgen`) sees it at -//! /mnt/9p/harness with zero configuration. `--unix`, `--tcp` and `--fd` -//! are the other listen forms; `--no-post` serves without posting; the -//! five harness roots default to $HOME/. and `--root NAME=PATH` -//! pins any of them elsewhere (the tests use fake homes). -//! -//! v1 has no daemonization: run it in a zmx session, the zmx way: -//! -//! zmx run harness -d 9harness -//! -//! SIGTERM or SIGINT stops it cleanly (a posted name is unposted). -const std = @import("std"); -const builtin = @import("builtin"); -const cloud9 = @import("cloud9"); -const tree = @import("tree.zig"); - -const fs = cloud9.fs; -const serve = cloud9.serve; -const post = cloud9.post; -const transport = cloud9.transport; -const Io = std.Io; -const linux = std.os.linux; - -const usage_text = - \\usage: 9harness [--unix PATH | --tcp IP:PORT | --fd N] [--no-post] - \\ [--name NAME] [--root NAME=PATH]... [--proc DIR] - \\ [--allow-move] [--zmx PATH] - \\ - \\A read-only, fresh-from-disk 9P2000 view of every harness's state: - \\ /pid /uptime daemon facts - \\ /claude/{projects,history,skills} - \\ /codex/{sessions,session-index,history} - \\ /omp /hermes /dsh full mirrors, raw - \\ /skills/{claude,codex,omp} the union skills view - \\ /active/// what is running right now - \\ - \\By default the daemon posts itself under the name `harness`, so it is - \\dialable at $XDG_RUNTIME_DIR/9p/harness and mountable by 9ns --mntgen - \\at /mnt/9p/harness. --no-post skips posting; --unix/--tcp add plain - \\listeners beside the post; --fd N serves one 9P session over the - \\connected stream on descriptor N and posts nothing. - \\--root NAME=PATH pins one harness root (NAME: claude, codex, omp, - \\hermes, dsh) somewhere other than $HOME/.; repeatable. - \\--proc DIR reads the process tree somewhere other than /proc (for - \\tests); --proc "" leaves /active out of the tree entirely. - \\--allow-move lets a write to /active///zmx move that agent - \\into a zmx session of that name: the daemon kills it and re-execs - \\the harness under zmx with its session resumed. Off by default, - \\because it is the one place the tree is not read-only, and anything - \\that can mount it can then kill an agent. --zmx PATH names the - \\binary a move runs (default: zmx, found on $PATH). - \\ - \\Run it in a zmx session, the zmx way: `zmx run harness -d 9harness`. - \\ -; - -const max_connections = 16; -const Runner = serve.Runner(tree.Harness, tree.opts, .{ - .msize = tree.msize, - .connections = max_connections, - .listeners = 2, // the posted name plus one of --unix/--tcp -}); - -// Static memory, the 9proc-demo way: the harness and the runner are large -// (the path table, the per-connection buffers) and live in .bss. -var harness_mem: tree.Harness = undefined; -var runner_mem: Runner = undefined; -var stop_requested = std.atomic.Value(bool).init(false); - -fn onSignal(sig: linux.SIG) callconv(.c) void { - _ = sig; - stop_requested.store(true, .release); -} - -fn installSignalHandlers() void { - const act = linux.Sigaction{ - .handler = .{ .handler = onSignal }, - .mask = @splat(0), - .flags = 0, - }; - _ = linux.sigaction(.TERM, &act, null); - _ = linux.sigaction(.INT, &act, null); - const ign = linux.Sigaction{ - .handler = .{ .handler = linux.SIG.IGN }, - .mask = @splat(0), - .flags = 0, - }; - _ = linux.sigaction(.PIPE, &ign, null); -} - -// ---- the serve handler --------------------------------------------------------- - -fn serveReq(ctx: ?*anyopaque, conn: *Runner.Conn, req: fs.Req) void { - const h: *tree.Harness = @ptrCast(@alignCast(ctx.?)); - // The answer's bytes point into the harness's shared buffers, so the - // mutex spans handle and reply: the engine copies them into the - // connection's output before the lock goes. - h.mutex.lockUncancelable(h.io); - const a = tree.handle(h, req); - conn.reply(&a.reply, a.bytes); - h.mutex.unlock(h.io); -} - -/// `name` found on `path_env`, as an absolute path in the arena. -fn onPath(io: Io, arena: std.mem.Allocator, path_env: []const u8, name: []const u8) ![]const u8 { - var it = std.mem.splitScalar(u8, path_env, ':'); - while (it.next()) |dir_path| { - if (dir_path.len == 0) continue; - const candidate = try std.fmt.allocPrint(arena, "{s}/{s}", .{ dir_path, name }); - const st = Io.Dir.statFile(.cwd(), io, candidate, .{ .follow_symlinks = true }) catch continue; - if (st.kind != .file) continue; - return candidate; - } - return error.NotFound; -} - -// ---- the CLI -------------------------------------------------------------------- - -const Mode = enum { posted, unix, tcp, fd }; - -pub fn main(init: std.process.Init) !void { - run(init) catch |e| switch (e) { - // Both are reported with their own line above; a returned error - // would bury it under a Debug-build stack trace. - error.Usage, error.AlreadyPosted => std.process.exit(1), - else => return e, - }; -} - -fn run(init: std.process.Init) !void { - const io = init.io; - const arena = init.arena.allocator(); - const args = try init.minimal.args.toSlice(arena); - const envp: post.Env = init.minimal.environ.block.slice.ptr; - - var mode: Mode = .posted; - var unix_path: []const u8 = ""; - var tcp_addr: []const u8 = ""; - var fd_no: i32 = -1; - var post_name: []const u8 = "harness"; - var no_post = false; - var base_overrides: [5]?[]const u8 = @splat(null); - var proc_root: []const u8 = "/proc"; - var zmx_path: []const u8 = "zmx"; - var allow_move = false; - - var i: usize = 1; - while (i < args.len) : (i += 1) { - const a = args[i]; - if (std.mem.eql(u8, a, "--no-post")) { - no_post = true; - } else if (std.mem.eql(u8, a, "--allow-move")) { - allow_move = true; - } else if (std.mem.eql(u8, a, "--unix") or std.mem.eql(u8, a, "--tcp") or - std.mem.eql(u8, a, "--fd") or std.mem.eql(u8, a, "--name") or - std.mem.eql(u8, a, "--proc") or std.mem.eql(u8, a, "--zmx") or - std.mem.eql(u8, a, "--root")) - { - i += 1; - if (i >= args.len) { - std.debug.print("9harness: {s} needs an argument\n{s}", .{ a, usage_text }); - return error.Usage; - } - const v = args[i]; - if (std.mem.eql(u8, a, "--unix")) { - mode = .unix; - unix_path = v; - } else if (std.mem.eql(u8, a, "--tcp")) { - mode = .tcp; - tcp_addr = v; - } else if (std.mem.eql(u8, a, "--fd")) { - mode = .fd; - fd_no = std.fmt.parseInt(i32, v, 10) catch { - std.debug.print("9harness: --fd: not a number: {s}\n", .{v}); - return error.Usage; - }; - } else if (std.mem.eql(u8, a, "--proc")) { - proc_root = v; - } else if (std.mem.eql(u8, a, "--zmx")) { - zmx_path = v; - } else if (std.mem.eql(u8, a, "--name")) { - post_name = v; - if (!post.legalName(post_name)) { - std.debug.print("9harness: illegal post name: {s}\n", .{v}); - return error.Usage; - } - } else { - const eq = std.mem.indexOfScalar(u8, v, '=') orelse { - std.debug.print("9harness: --root NAME=PATH: {s}\n{s}", .{ v, usage_text }); - return error.Usage; - }; - const name = v[0..eq]; - const root = std.meta.stringToEnum(tree.Root, name) orelse { - std.debug.print("9harness: unknown root {s} (claude, codex, omp, hermes, dsh)\n", .{name}); - return error.Usage; - }; - base_overrides[@intFromEnum(root)] = v[eq + 1 ..]; - } - } else if (std.mem.eql(u8, a, "--help") or std.mem.eql(u8, a, "-h")) { - std.debug.print("{s}", .{usage_text}); - return; - } else { - std.debug.print("9harness: unknown argument {s}\n{s}", .{ a, usage_text }); - return error.Usage; - } - } - - // The five pinned roots: $HOME/. unless overridden. - const home = post.getenv(envp, "HOME"); - var bases: [5][]const u8 = @splat(""); - inline for (0..5) |r| { - if (base_overrides[r]) |path| { - bases[r] = path; - } else if (home) |hm| { - bases[r] = std.fmt.bufPrint(&root_bufs[r], "{s}/.{s}", .{ hm, root_dirs[r] }) catch return error.NameTooLong; - } - } - var any = false; - for (bases) |b| any = any or b.len != 0; - if (!any) { - std.debug.print("9harness: no harness roots (set $HOME or pass --root NAME=PATH)\n", .{}); - return error.Usage; - } - - // A move execs zmx by absolute path: the daemon resolves it once, - // at startup, so nothing about $PATH matters when the write lands. - const zmx_abs = if (std.mem.indexOfScalar(u8, zmx_path, '/') != null) - zmx_path - else - onPath(io, arena, post.getenv(envp, "PATH") orelse "", zmx_path) catch zmx_path; - if (allow_move and std.mem.indexOfScalar(u8, zmx_abs, '/') == null) { - std.debug.print("9harness: --allow-move needs zmx on $PATH (or --zmx PATH)\n", .{}); - return error.Usage; - } - - harness_mem.init(.{ - .io = io, - .pid = @intCast(linux.getpid()), - .bases = bases, - .proc = proc_root, - .home = home orelse "", - .zmx = zmx_abs, - .runtime = post.getenv(envp, "XDG_RUNTIME_DIR") orelse "", - .allow_move = allow_move, - .envp = envp, - }); - if (allow_move) { - std.debug.print("9harness: moves allowed — a write to /active///zmx re-execs that agent under {s}\n", .{zmx_abs}); - } - if (no_post and mode == .posted) { - std.debug.print("9harness: --no-post needs a listen form (--unix, --tcp or --fd)\n{s}", .{usage_text}); - return error.Usage; - } - - installSignalHandlers(); - - switch (mode) { - .fd => { - std.debug.print("9harness: serving one session on fd {d}\n", .{fd_no}); - try serveFd(io, fd_no); - return; - }, - .posted, .unix, .tcp => {}, - } - - runner_mem.init(.{ .io = io, .root = tree.root, .handler = .{ .ctx = &harness_mem, .serve = serveReq } }); - defer runner_mem.stop(); - - var posted_something = false; - if (!no_post) { - runner_mem.listenPosted(envp, post_name, 16) catch |err| { - if (err == error.AlreadyPosted) { - std.debug.print("9harness: the name `{s}` is already posted by a live server\n", .{post_name}); - return err; - } - std.debug.print("9harness: cannot post as {s}: {t}\n", .{ post_name, err }); - return err; - }; - posted_something = true; - var pbuf: [transport.sun_path_len]u8 = undefined; - const path = post.registryPath(envp, post_name, &pbuf) catch ""; - std.debug.print("9harness: posted as {s} at {s}\n", .{ post_name, path }); - } - switch (mode) { - .unix => { - _ = try runner_mem.listen(.{ .unix = try arena.dupeZ(u8, unix_path) }, 16); - std.debug.print("9harness: listening on {s}\n", .{unix_path}); - }, - .tcp => { - const addr = Io.net.IpAddress.parseLiteral(tcp_addr) catch { - std.debug.print("9harness: bad --tcp address: {s}\n", .{tcp_addr}); - return error.Usage; - }; - const bound = try runner_mem.listen(.{ .tcp = addr }, 16); - std.debug.print("9harness: listening on tcp!{f}\n", .{bound}); - }, - else => {}, - } - var pinned: u32 = 0; - for (bases) |b| { - if (b.len != 0) pinned += 1; - } - std.debug.print("9harness: serving ({d} roots pinned, {d} bytes of tables)\n", .{ pinned, @sizeOf(tree.Harness) }); - - // The runner's tasks drive themselves; the main thread only waits for - // a stop signal (SIGTERM/SIGINT) and then unposts via stop(). - while (!stop_requested.load(.acquire)) { - io.sleep(.fromMilliseconds(250), .awake) catch break; - } - std.debug.print("9harness: stopping\n", .{}); -} - -const root_dirs = [5][]const u8{ "claude", "codex", "omp", "hermes", "dsh" }; -var root_bufs: [5][tree.base_capacity]u8 = @splat(@splat(0)); - -// ---- --fd N: one 9P session over a connected stream ----------------------------- - -/// Serves exactly one client on the connected stream of descriptor `n` -/// (socket activation, `9ns --spawn`-style handoffs), driving the engine -/// directly, the freestanding way. Returns when the client hangs up. -fn serveFd(io: Io, n: i32) !void { - const stream: Io.net.Stream = .{ .socket = .{ .handle = n, .address = .{ .ip4 = .loopback(0) } } }; - const Engine = fs.Server(tree.Harness, tree.opts); - var in: [tree.msize]u8 = undefined; - var out: [2 * tree.msize]u8 = undefined; - var rbuf: [tree.msize]u8 = undefined; - var wbuf: [2 * tree.msize]u8 = undefined; - var stage: [tree.msize]u8 = undefined; - var engine = Engine.init(.{ .in = &in, .out = &out, .root = tree.root }); - var reader = stream.reader(io, &rbuf); - var writer = stream.writer(io, &wbuf); - while (true) { - const frame = transport.readFrame(&reader.interface, &stage, tree.msize) catch return; - var off: usize = 0; - while (off < frame.len) { - off += engine.push(frame[off..]); - while (true) { - const req = engine.next() orelse break; - const a = answer(req); - engine.reply(&a.reply, a.bytes); - } - const pending = engine.output(); - writer.interface.writeAll(pending) catch return; - engine.wrote(pending.len); - if (engine.protocol.dead) return; - } - writer.interface.flush() catch return; - } -} - -/// One request for the --fd loop: same locking discipline as the runner's -/// handler (the reply bytes live in the harness's shared buffers). -fn answer(req: fs.Req) tree.Answer { - harness_mem.mutex.lockUncancelable(harness_mem.io); - defer harness_mem.mutex.unlock(harness_mem.io); - return tree.handle(&harness_mem, req); -} - -test { - _ = @import("tree.zig"); // the tree's unit tests (exclusions, ids, ...) -} - -test "main: the root override parser pins each named root" { - // Parsing is inline in run(); the mapping itself is what the tests - // rely on, and it is exercised end to end in test/e2e.sh. - _ = std.meta.stringToEnum(tree.Root, "claude").?; - _ = std.meta.stringToEnum(tree.Root, "codex").?; - _ = std.meta.stringToEnum(tree.Root, "omp").?; - _ = std.meta.stringToEnum(tree.Root, "hermes").?; - _ = std.meta.stringToEnum(tree.Root, "dsh").?; - try std.testing.expect(std.meta.stringToEnum(tree.Root, "zmx") == null); -} diff --git a/9harness/src/tree.zig b/9harness/src/tree.zig deleted file mode 100644 index b6ff4a4..0000000 --- a/9harness/src/tree.zig +++ /dev/null @@ -1,2197 +0,0 @@ -//! The harness file tree served over 9P2000: a unified, read-only, -//! fresh-from-disk view of every AI-agent harness's state on the machine. -//! -//! /pid /uptime daemon facts, one line each -//! /claude/ projects/ (mirror of ~/.claude/projects), -//! history (~/.claude/history.jsonl), -//! skills/ (mirror of ~/.claude/skills) -//! /codex/ sessions/ (~/.codex/sessions), -//! session-index (~/.codex/session_index.jsonl), -//! history (~/.codex/history.jsonl) -//! /omp/ mirror of ~/.omp/agent, raw blobs -//! /hermes/ mirror of ~/.hermes, raw -//! /dsh/ mirror of ~/.dsh -//! /skills/{claude,codex,omp}/ the same skills trees the harnesses own -//! -//! Everything below the named mount points is a lazy mirror: a lookup, -//! getattr or readdir walks the real filesystem at request time, so a -//! session transcript grows as its harness writes it and a new session -//! appears as soon as its file lands. No cache, no invalidation. -//! -//! Security boundary — the exclusion rule is absolute and unit-tested: -//! no name that looks like a credential, key, token or auth store is ever -//! answered, at any depth (`excluded`); symlinks are never served (they -//! are an escape hatch around the pinned roots). Every path is resolved -//! from its pinned root one component at a time with `O_NOFOLLOW` -//! (`openIn`), so no name below a root — swapped mid-session or not — -//! can point the daemon at a file outside it. The daemon must never -//! become a credential reader for anything that mounts it. Writes, -//! creates and setattrs answer EPERM. -//! -//! This module is the backend of `cloud9.fs.Server` (main.zig hands it to -//! `serve.Runner`). It declares `features = .{ .references = true }`: every -//! lookup result is a reference, so table entries below (the mirrored -//! files) are refcounted and freed when the last fid lets go. Node ids -//! pack (kind, root index, serial) in a u64, zmx-style: a mirrored file's -//! serial is its table slot and generation, so two walks of the same file -//! answer the same qid path while it is held. -const std = @import("std"); -const cloud9 = @import("cloud9"); -const fs = cloud9.fs; -const E = fs.E; -const Io = std.Io; -const linux = std.os.linux; -pub const active = @import("active.zig"); - -// ---- comptime bounds --------------------------------------------------------- - -/// Longest relative path a mirrored file may carry (bytes below a root). -pub const rel_capacity: usize = 640; -/// Mirrored files remembered at once; each holds one reference per fid. -pub const path_capacity: usize = 4096; -comptime { - if (path_capacity > 1 << 12) @compileError("path table slots must fit the 12 serial bits"); -} -/// Largest frame the daemon negotiates; read and readdir staging buffers -/// are this big, so a single Rread can carry one full frame of bytes. -pub const msize: u32 = 32 * 1024; -/// Names one directory listing may stage (sorted for cursor stability). -pub const list_capacity: usize = 1024; -/// Bytes of name storage one listing may use (255 per name, packed). -pub const list_name_bytes: usize = 128 * 1024; -/// Longest base path a pinned root may carry. -pub const base_capacity: usize = 512; - -pub const opts: fs.Options = .{ - .fid_capacity = 512, - .slot_capacity = 8, - .name_capacity = 255, - .username_capacity = 28, - .fid_index = true, -}; - -// ---- the tree skeleton ------------------------------------------------------ - -pub const Root = enum(u8) { claude, codex, omp, hermes, dsh }; - -/// Static nodes: the facts, the harness dirs and the union skills dir. -pub const Top = enum(u8) { - root = 1, - pid, - uptime, - claude, - codex, - omp, - hermes, - dsh, - skills, - active, - - /// Directory entries of the root, in listing order. - pub const listed = [_]Top{ .pid, .uptime, .claude, .codex, .omp, .hermes, .dsh, .skills, .active }; - - pub fn fileName(t: Top) []const u8 { - return @tagName(t); - } - - pub fn dir(t: Top) bool { - return switch (t) { - .root, .claude, .codex, .omp, .hermes, .dsh, .skills, .active => true, - .pid, .uptime => false, - }; - } - - fn parent(t: Top) Top { - _ = t; - return .root; // the root is its own parent; every top hangs off it - } - - /// The mirror a harness dir serves: its root and the relative path - /// below it. The claude and codex dirs are *virtual* (their children - /// are named mounts); omp, hermes and dsh mirror one subtree each. - fn mirror(t: Top) ?Mount { - return switch (t) { - .omp => .{ .root = .omp, .rel = "agent" }, - .hermes => .{ .root = .hermes, .rel = "" }, - .dsh => .{ .root = .dsh, .rel = "" }, - else => null, - }; - } -}; - -/// Where a mirrored subtree sits: `rel` below the root's pinned base path. -pub const Mount = struct { - root: Root, - rel: []const u8, -}; - -const NamedMount = struct { - /// The name as it appears in its virtual directory. - name: []const u8, - owner: Top, - mount: Mount, -}; - -/// The children of the virtual directories (claude, codex, skills). The -/// same target may appear twice (/claude/skills and /skills/claude): a -/// lookup of either hands out the same table entry, so both paths share -/// qid paths and identity. -const named_mounts = [_]NamedMount{ - .{ .name = "projects", .owner = .claude, .mount = .{ .root = .claude, .rel = "projects" } }, - .{ .name = "history", .owner = .claude, .mount = .{ .root = .claude, .rel = "history.jsonl" } }, - .{ .name = "skills", .owner = .claude, .mount = .{ .root = .claude, .rel = "skills" } }, - .{ .name = "sessions", .owner = .codex, .mount = .{ .root = .codex, .rel = "sessions" } }, - .{ .name = "session-index", .owner = .codex, .mount = .{ .root = .codex, .rel = "session_index.jsonl" } }, - .{ .name = "history", .owner = .codex, .mount = .{ .root = .codex, .rel = "history.jsonl" } }, - .{ .name = "claude", .owner = .skills, .mount = .{ .root = .claude, .rel = "skills" } }, - .{ .name = "codex", .owner = .skills, .mount = .{ .root = .codex, .rel = "skills" } }, - .{ .name = "omp", .owner = .skills, .mount = .{ .root = .omp, .rel = "skills" } }, -}; - -fn mountNamed(owner: Top, name: []const u8) ?Mount { - for (named_mounts) |m| { - if (m.owner == owner and std.mem.eql(u8, m.name, name)) return m.mount; - } - return null; -} - -/// The Top that owns the mount whose target is exactly (root, rel) — the -/// canonical parent a ".." walk answers. Falls back to the root's harness -/// dir for targets no named mount covers (deep paths, mirrors). -fn mountOwner(r: Root, rel_path: []const u8) Top { - for (named_mounts) |m| { - if (m.mount.root == r and std.mem.eql(u8, m.mount.rel, rel_path)) return m.owner; - } - return switch (r) { - .claude => .claude, - .codex => .codex, - .omp => .omp, - .hermes => .hermes, - .dsh => .dsh, - }; -} - -// ---- node ids and the path table --------------------------------------------- - -pub const Kind = enum(u8) { top = 0, path = 1, active = 2 }; - -comptime { - // The derived view indexes the same pinned roots the mirror does. - for (std.enums.values(Root)) |r| { - if (!std.mem.eql(u8, @tagName(r), @tagName(@as(active.Kind, @enumFromInt(@intFromEnum(r)))))) - @compileError("tree.Root and active.Kind must agree, index by index"); - } -} - -/// A node under `/active`. Everything but a harness directory names a -/// slot of the live table and the generation it had when the id was -/// minted, so an id outlives its process only long enough to answer -/// ENOENT. -pub const AFile = enum(u8) { - /// `/active/` - harness_dir, - /// `/active//` - entry, - pid, - ppid, - started, - cwd, - name, - title, - session, - model, - via, - zmx, - status, - transcript, - /// `/active///agents` - agents, - /// `/active///agents/` - agent, - agent_model, - agent_transcript, - - /// The fields of one entry, in listing order. A field with no value - /// is not listed and does not resolve. - pub const entry_files = [_]AFile{ - .pid, .ppid, .started, .cwd, .name, .title, - .session, .model, .via, .zmx, .status, .transcript, - }; - pub const agent_files = [_]AFile{ .agent_model, .agent_transcript }; - - pub fn fileName(f: AFile) []const u8 { - return switch (f) { - .agent_model => "model", - .agent_transcript => "transcript", - else => @tagName(f), - }; - } -}; - -/// A resolved `/active` node. -pub const Act = struct { - file: AFile, - /// Only for `harness_dir`. - kind: active.Kind = .claude, - slot: u8 = 0, - gen: u8 = 0, - agent: u8 = 0, -}; - -pub fn activeNode(a: Act) u64 { - const serial: u48 = switch (a.file) { - .harness_dir => @intFromEnum(a.kind), - else => @as(u48, a.slot) | (@as(u48, a.gen) << 8) | (@as(u48, a.agent) << 16), - }; - return @bitCast(Node{ - .idx = @intFromEnum(a.file), - .kind = @intFromEnum(Kind.active), - .serial = serial, - }); -} - -pub const Node = packed struct(u64) { - /// A Top index (kind == .top) or a Root index (kind == .path). - idx: u8 = 0, - kind: u8 = 0, - serial: u48 = 0, -}; - -pub fn topNode(t: Top) u64 { - return @bitCast(Node{ .idx = @intFromEnum(t), .kind = @intFromEnum(Kind.top) }); -} - -pub const root: u64 = topNode(Top.root); - -/// One remembered mirrored file: its root, its relative path, the hash -/// that short-circuits dedupe lookups, and its reference count (one per -/// fid holding it, handed out by lookup, paid back by release). -const Entry = struct { - used: bool = false, - root: Root = .claude, - rel_buf: [rel_capacity]u8 = undefined, - rel_len: u16 = 0, - hash: u64 = 0, - refs: u32 = 0, - gen: u32 = 0, - - fn rel(e: *const Entry) []const u8 { - return e.rel_buf[0..e.rel_len]; - } -}; - -fn serialOf(slot: u12, gen: u32) u48 { - return (@as(u48, gen) << 12) | slot; -} - -fn nodeOf(e: *const Entry, slot: u12) u64 { - return @bitCast(Node{ - .idx = @intFromEnum(e.root), - .kind = @intFromEnum(Kind.path), - .serial = serialOf(slot, e.gen), - }); -} - -// ---- the exclusions (the security boundary) ----------------------------------- - -fn containsFold(name: []const u8, needle: []const u8) bool { - if (name.len < needle.len) return false; - var i: usize = 0; - while (i + needle.len <= name.len) : (i += 1) { - if (std.ascii.eqlIgnoreCase(name[i .. i + needle.len], needle)) return true; - } - return false; -} - -fn endsWithFold(name: []const u8, suffix: []const u8) bool { - return name.len >= suffix.len and std.ascii.eqlIgnoreCase(name[name.len - suffix.len ..], suffix); -} - -fn isOneOf(name: []const u8, comptime names: []const []const u8) bool { - inline for (names) |n| if (std.ascii.eqlIgnoreCase(name, n)) return true; - return false; -} - -/// Absolute rule: a name that may hold a credential, key, token or auth -/// material is never served, at any depth, by lookup or readdir. The -/// substrings are folded (case-insensitive) and deliberately broad — -/// "auth" also hides "author-notes", the price of never guessing wrong. -/// When unsure, exclude and document (docs/DESIGN.md). -pub fn excluded(name: []const u8) bool { - @setEvalBranchQuota(20000); - if (name.len == 0) return true; - for ([_][]const u8{ "credentials", "token", "auth", "secret" }) |needle| { - if (containsFold(name, needle)) return true; - } - if (endsWithFold(name, ".key")) return true; - if (endsWithFold(name, ".pem")) return true; - if (endsWithFold(name, ".env")) return true; - // Config and settings files of the other harnesses may embed API keys - // (hermes config.yaml, omp config.yml, dsh settings.yaml). Claude - // Code's settings.json is not in the tree at all: /claude serves only - // projects/, skills/ and history.jsonl. - if (isOneOf(name, &.{ "settings.json", "settings.yaml", "settings.local.json", "config.yml", "config.yaml", "config.toml", "models.yml", "models.yaml", ".ssh" })) return true; - for ([_][]const u8{ "id_rsa", "id_dsa", "id_ecdsa", "id_ed25519" }) |prefix| { - if (name.len >= prefix.len and std.ascii.eqlIgnoreCase(name[0..prefix.len], prefix)) return true; - } - return false; -} - -/// Every component of a relative path must pass `excluded`; used on the -/// comptime mount targets and the unit-test fixtures alike. -pub fn excludedPath(rel_path: []const u8) bool { - @setEvalBranchQuota(4000); - var it = std.mem.splitScalar(u8, rel_path, '/'); - while (it.next()) |comp| { - if (excluded(comp)) return true; - } - return false; -} - -// ---- the backend -------------------------------------------------------------- - -pub const Answer = struct { - reply: fs.Reply, - bytes: []const u8 = "", -}; - -fn fail(tag: u64, e: u16) Answer { - return .{ .reply = fs.Reply.fail(tag, e) }; -} - -/// The daemon's state. One instance serves every connection; `handle` is -/// called with `mutex` held (the serve handler in main.zig holds it across -/// handle and reply, because `bytes` point into the shared buffers). -pub const Harness = struct { - pub const Req = fs.Req; - pub const Reply = fs.Reply; - pub const features: fs.Features = .{ .references = true }; - - io: Io, - pid: u32 = 0, - started_sec: i64 = 0, - /// The five pinned roots; a root without a base is unreachable. - base_buf: [5][base_capacity]u8 = @splat(@splat(0)), - base_len: [5]u16 = @splat(0), - base_set: [5]bool = @splat(false), - table: [path_capacity]Entry = @splat(.{}), - mutex: Io.Mutex = .init, - // Shared staging (guarded by `mutex`). - data_buf: [msize]u8 = undefined, - stage_buf: [msize]u8 = undefined, - list_names: [list_name_bytes]u8 = undefined, - list_dirs: [list_capacity]bool = undefined, - list_offs: [list_capacity]u32 = undefined, - - // The derived view (`/active`). An empty `proc` base turns it off, - // which is the default: a test must opt in to reading a process - // tree, and never the live one. - live: active.Table = .{}, - proc_buf: [base_capacity]u8 = @splat(0), - proc_len: u16 = 0, - home_buf: [base_capacity]u8 = @splat(0), - home_len: u16 = 0, - zmx_buf: [base_capacity]u8 = @splat(0), - zmx_len: u16 = 0, - runtime_buf: [base_capacity]u8 = @splat(0), - runtime_len: u16 = 0, - envp: ?cloud9.post.Env = null, - self_pid: u32 = 0, - btime: i64 = 0, - allow_move: bool = false, - act_text: [1024]u8 = undefined, - act_name: [64]u8 = undefined, - act_path: [active.path_capacity]u8 = undefined, - - pub const InitOptions = struct { - io: Io, - pid: u32, - /// Base path of each root, or "" when the root is not pinned. - bases: [5][]const u8, - /// The proc filesystem `/active` reads. Empty — the default — - /// leaves the derived view out of the tree entirely. - proc: []const u8 = "", - /// $HOME, for the harnesses' store-slug spellings. - home: []const u8 = "", - /// The zmx binary a move execs, resolved to an absolute path by - /// the caller. Never client bytes. - zmx: []const u8 = "zmx", - /// $XDG_RUNTIME_DIR, where a move looks for the zmx session it - /// is about to create. - runtime: []const u8 = "", - /// The daemon's own environment block, handed to the harness a - /// move re-execs. Null leaves a moved agent with an empty one. - envp: ?cloud9.post.Env = null, - /// Whether writing `/active///zmx` may move a session. - allow_move: bool = false, - }; - - pub fn init(h: *Harness, o: InitOptions) void { - h.* = .{ .io = o.io, .pid = o.pid, .started_sec = Io.Timestamp.now(o.io, .real).toSeconds() }; - if (o.proc.len > 0 and o.proc.len <= base_capacity) { - @memcpy(h.proc_buf[0..o.proc.len], o.proc); - h.proc_len = @intCast(o.proc.len); - h.self_pid = o.pid; - h.btime = active.bootTime(o.io, o.proc); - } - if (o.home.len <= base_capacity) { - @memcpy(h.home_buf[0..o.home.len], o.home); - h.home_len = @intCast(o.home.len); - } - if (o.zmx.len > 0 and o.zmx.len <= base_capacity) { - @memcpy(h.zmx_buf[0..o.zmx.len], o.zmx); - h.zmx_len = @intCast(o.zmx.len); - } - if (o.runtime.len <= base_capacity) { - @memcpy(h.runtime_buf[0..o.runtime.len], o.runtime); - h.runtime_len = @intCast(o.runtime.len); - } - h.allow_move = o.allow_move; - h.envp = o.envp; - inline for (0..5) |i| { - const path = o.bases[i]; - h.base_set[i] = path.len > 0 and path.len <= base_capacity; - if (h.base_set[i]) { - @memcpy(h.base_buf[i][0..path.len], path); - h.base_len[i] = @intCast(path.len); - } - } - } - - pub fn base(h: *const Harness, r: Root) ?[]const u8 { - if (!h.base_set[@intFromEnum(r)]) return null; - return h.base_buf[@intFromEnum(r)][0..h.base_len[@intFromEnum(r)]]; - } - - // -- path table -- - - fn hashOf(r: Root, rel_path: []const u8) u64 { - var hh = std.hash.Wyhash.init(0x6861_726e_6573_7300); // "harness" - hh.update(&.{@intFromEnum(r)}); - hh.update(rel_path); - return hh.final(); - } - - /// Finds or creates the entry for (root, rel). `ref` says whether the - /// caller hands out a reference (a lookup) or only names the file (a - /// readdir record). Two walks of the same file find the same entry, - /// so their node ids are equal; a freed entry's generation changes. - fn entryFor(h: *Harness, r: Root, rel_path: []const u8, ref: bool) ?*Entry { - if (rel_path.len == 0 or rel_path.len > rel_capacity) return null; - const hash = hashOf(r, rel_path); - var free: ?*Entry = null; - for (&h.table) |*e| { - if (!e.used) { - if (free == null) free = e; - continue; - } - if (e.hash == hash and e.root == r and e.rel_len == rel_path.len and - std.mem.eql(u8, e.rel(), rel_path)) - { - if (ref) e.refs += 1; - return e; - } - } - const e = free orelse return null; - e.used = true; - e.root = r; - e.rel_len = @intCast(rel_path.len); - @memcpy(e.rel_buf[0..rel_path.len], rel_path); - e.hash = hash; - e.refs = @intFromBool(ref); - e.gen +%= 1; - return e; - } - - fn slotOf(h: *Harness, e: *Entry) u12 { - const off = (@intFromPtr(e) - @intFromPtr(&h.table)) / @sizeOf(Entry); - return @intCast(off); - } - - /// The target a node names; a freed or forged serial answers null. - fn resolve(h: *Harness, node: u64) ?Target { - const n: Node = @bitCast(node); - switch (std.enums.fromInt(Kind, n.kind) orelse return null) { - .top => return .{ .top = std.enums.fromInt(Top, n.idx) orelse return null }, - .path => { - const r = std.enums.fromInt(Root, n.idx) orelse return null; - const slot: u12 = @truncate(n.serial); - const e = &h.table[slot]; - if (!e.used or e.root != r) return null; - if (serialOf(slot, e.gen) != n.serial) return null; - return .{ .path = e }; - }, - .active => { - const f = std.enums.fromInt(AFile, n.idx) orelse return null; - if (f == .harness_dir) { - const k = std.enums.fromInt(active.Kind, @as(u8, @truncate(n.serial))) orelse return null; - return .{ .act = .{ .file = f, .kind = k } }; - } - const slot: u8 = @truncate(n.serial); - const gen: u8 = @truncate(n.serial >> 8); - const agent: u8 = @truncate(n.serial >> 16); - const l = h.live.at(slot, gen) orelse return null; - return .{ .act = .{ .file = f, .kind = l.kind, .slot = slot, .gen = gen, .agent = agent } }; - }, - } - } - - fn entryNode(h: *Harness, e: *Entry) u64 { - return nodeOf(e, h.slotOf(e)); - } -}; - -pub const Target = union(enum) { - top: Top, - path: *Entry, - act: Act, -}; - -// ---- attributes --------------------------------------------------------------- - -/// Opens `/` without following a symlink at any step -/// below the base: every component is opened with `O_NOFOLLOW`, so a -/// name swapped for a symlink between the walk and the read fails -/// instead of reaching out of the root. Composing the whole path and -/// opening it in one call cannot do that — only the last component's -/// symlink is refused then, and every directory above it is followed -/// wherever it points. The base itself is configuration (`--root` or -/// $HOME), not client bytes, so it resolves normally. The caller owns -/// the descriptor. -fn openIn(h: *Harness, r: Root, rel_path: []const u8, final: linux.O) ?i32 { - const b = h.base(r) orelse return null; - if (rel_path.len > rel_capacity or b.len > base_capacity) return null; - const walk: linux.O = .{ .PATH = true, .DIRECTORY = true, .CLOEXEC = true, .NOFOLLOW = true }; - - var base_z: [base_capacity + 1]u8 = @splat(0); - @memcpy(base_z[0..b.len], b); - var top_flags = if (rel_path.len == 0) final else walk; - top_flags.NOFOLLOW = false; // the pinned root may legitimately be a link - top_flags.CLOEXEC = true; - var fd = fdOf(linux.open(@ptrCast(&base_z), top_flags, 0)) orelse return null; - if (rel_path.len == 0) return fd; - - var rest = rel_path; - while (true) { - const slash = std.mem.indexOfScalar(u8, rest, '/'); - const comp = if (slash) |i| rest[0..i] else rest; - const last = slash == null; - // Nothing below composes these; a component that could re-enter - // the walk is refused rather than opened. - if (comp.len == 0 or comp.len > 255 or - std.mem.eql(u8, comp, ".") or std.mem.eql(u8, comp, "..")) - { - _ = linux.close(fd); - return null; - } - var name_z: [256]u8 = @splat(0); - @memcpy(name_z[0..comp.len], comp); - var flags = if (last) final else walk; - flags.NOFOLLOW = true; - flags.CLOEXEC = true; - const next = fdOf(linux.openat(fd, @ptrCast(&name_z), flags, 0)); - _ = linux.close(fd); - fd = next orelse return null; - if (last) return fd; - rest = rest[comp.len + 1 ..]; - } -} - -fn fdOf(rc: usize) ?i32 { - if (linux.errno(rc) != .SUCCESS) return null; - return @intCast(rc); -} - -/// Stat of `/`, resolved the no-follow way. A symlink -/// is never served: it is the one escape hatch around the pinned roots. -/// `O_PATH` opens the name itself, so a fifo or device never blocks and -/// never has its open side effects run. -fn statIn(h: *Harness, r: Root, rel_path: []const u8) ?Io.Dir.Stat { - const fd = openIn(h, r, rel_path, .{ .PATH = true, .CLOEXEC = true }) orelse return null; - const f: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; - defer f.close(h.io); - const st = f.stat(h.io) catch return null; - if (st.kind == .sym_link) return null; // never served: an escape hatch - return st; -} - -fn statAttr(h: *Harness, e: *Entry) ?fs.Attr { - const st = statIn(h, e.root, e.rel()) orelse return null; - return .{ - .name = basename(e.rel()), - .node = h.entryNode(e), - .dir = st.kind == .directory, - .size = if (st.kind == .directory) 0 else st.size, - .mode = if (st.kind == .directory) 0o555 else 0o444, - .mtime = @truncate(@as(u64, @bitCast(st.mtime.toSeconds()))), - }; -} - -fn basename(rel_path: []const u8) []const u8 { - if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| return rel_path[i + 1 ..]; - return rel_path; -} - -/// A fact's text, rendered on demand. `buf` should be a small stack buffer. -fn factText(h: *Harness, fact: Top, buf: []u8) []const u8 { - var w = Io.Writer.fixed(buf); - switch (fact) { - .pid => w.print("{d}\n", .{h.pid}) catch {}, - .uptime => { - const now = Io.Timestamp.now(h.io, .real).toSeconds(); - const up: u64 = if (now > h.started_sec) @intCast(now - h.started_sec) else 0; - w.print("{d}\n", .{up}) catch {}; - }, - else => {}, - } - return w.buffered(); -} - -var fact_buf: [64]u8 = undefined; - -fn attrFor(h: *Harness, t: Target) ?fs.Attr { - switch (t) { - .top => |top| { - switch (top) { - .root => return .{ .name = "/", .node = root, .dir = true, .mode = 0o555 }, - .active => return .{ .name = "active", .node = topNode(.active), .dir = true, .mode = 0o555 }, - .pid, .uptime => return .{ - .name = top.fileName(), - .node = topNode(top), - .size = factText(h, top, &fact_buf).len, - .mode = 0o444, - }, - .claude, .codex, .skills => return .{ - .name = top.fileName(), - .node = topNode(top), - .dir = true, - .mode = 0o555, - }, - .omp, .hermes, .dsh => { - // The harness dir IS its mirror: stat the target. - const m = top.mirror().?; - const st = statIn(h, m.root, m.rel) orelse return null; - if (st.kind != .directory) return null; - return .{ .name = top.fileName(), .node = topNode(top), .dir = true, .mode = 0o555 }; - }, - } - }, - .path => |e| return statAttr(h, e), - .act => |a| return activeAttr(h, a), - } -} - -/// Attr of the mount target as a table entry (fresh from disk, or null -/// when the target is missing, a symlink, or excluded by name). -fn attrOfMount(h: *Harness, r: Root, rel_path: []const u8) ?fs.Attr { - const e = h.entryFor(r, rel_path, false) orelse return null; - defer if (e.refs == 0) { - e.used = false; - e.gen +%= 1; - }; - return statAttr(h, e); -} - -// ---- /active: the derived view ------------------------------------------------- - -fn sourcesOf(h: *Harness) active.Sources { - var roots: [5][]const u8 = @splat(""); - for (0..5) |i| { - if (h.base_set[i]) roots[i] = h.base_buf[i][0..h.base_len[i]]; - } - return .{ - .io = h.io, - .proc = h.proc_buf[0..h.proc_len], - .home = h.home_buf[0..h.home_len], - .roots = roots, - .self_pid = h.self_pid, - .btime = h.btime, - }; -} - -/// Rescans the process tree. Done on a listing of `/active` and on a -/// lookup that misses, which is what makes an agent visible the moment -/// it starts and gone the moment it exits. -fn rescan(h: *Harness) void { - if (h.proc_len == 0) return; - h.live.scan(sourcesOf(h)); -} - -fn hasLive(h: *Harness, k: active.Kind) bool { - for (&h.live.slots) |*sl| { - if (sl.used and sl.live.kind == k) return true; - } - return false; -} - -/// `\n` staged where an answer can point at it. -fn line(h: *Harness, text: []const u8) ?[]const u8 { - return std.fmt.bufPrint(&h.act_text, "{s}\n", .{text}) catch null; -} - -/// The value of a field file, or null when this agent has none — an -/// absent field is not listed and does not resolve, so the tree never -/// answers a blank where it does not know. -fn activeText(h: *Harness, a: Act) ?[]const u8 { - const l = h.live.at(a.slot, a.gen) orelse return null; - const src = sourcesOf(h); - var scratch: [active.text_capacity]u8 = undefined; - return switch (a.file) { - .pid => std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.pid}) catch null, - .ppid => if (l.ppid == 0) null else std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.ppid}) catch null, - .started => if (l.started == 0) null else std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.started}) catch null, - .cwd => if (l.cwd.len == 0) null else line(h, l.cwd.slice()), - .session => if (l.session.len == 0) null else line(h, l.session.slice()), - .via => line(h, l.via.text()), - .name => line(h, active.registryField(src, l, "name", &scratch) orelse return null), - .status => line(h, active.registryField(src, l, "status", &scratch) orelse return null), - .title => line(h, active.headField(h.io, l.kind, l.transcript.slice(), .title, &scratch) orelse return null), - .model => line(h, active.headField(h.io, l.kind, l.transcript.slice(), .model, &scratch) orelse return null), - .agent_model => blk: { - const path = activePath(h, a) orelse break :blk null; - break :blk line(h, active.headField(h.io, l.kind, path, .model, &scratch) orelse return null); - }, - // `zmx` is always there, so it can always be written to; empty - // means the agent runs outside zmx. - .zmx => if (active.zmxOf(src, l.pid, &scratch)) |v| line(h, v) else h.act_text[0..0], - else => null, - }; -} - -/// The absolute path behind a transcript-shaped file. -fn activePath(h: *Harness, a: Act) ?[]const u8 { - const l = h.live.at(a.slot, a.gen) orelse return null; - switch (a.file) { - .transcript => return if (l.transcript.len == 0) null else l.transcript.slice(), - .agent_transcript, .agent_model => { - var ag: active.Agents = .{}; - active.agentsOf(h.io, l, &ag); - return active.agentPath(l, &ag, a.agent, &h.act_path); - }, - else => return null, - } -} - -/// Opens an absolute path without following a symlink at the last -/// component. These paths are derived from a pinned root or from the -/// fd the harness itself holds, never from client bytes, but a name -/// swapped underneath must still fail rather than redirect. -fn openAbsNoFollow(path: []const u8) ?i32 { - if (path.len == 0 or path.len >= active.path_capacity) return null; - var z: [active.path_capacity]u8 = @splat(0); - @memcpy(z[0..path.len], path); - const rc = linux.open(@ptrCast(&z), .{ - .ACCMODE = .RDONLY, - .NOFOLLOW = true, - .CLOEXEC = true, - .NONBLOCK = true, - }, 0); - if (linux.errno(rc) != .SUCCESS) return null; - return @intCast(rc); -} - -fn activeAttr(h: *Harness, a: Act) ?fs.Attr { - switch (a.file) { - .harness_dir => { - if (!hasLive(h, a.kind)) return null; - return .{ .name = a.kind.text(), .node = activeNode(a), .dir = true, .mode = 0o555 }; - }, - .entry => { - const l = h.live.at(a.slot, a.gen) orelse return null; - const nm = std.fmt.bufPrint(&h.act_name, "{d}", .{l.pid}) catch return null; - return .{ .name = nm, .node = activeNode(a), .dir = true, .mode = 0o555 }; - }, - .agents => { - const l = h.live.at(a.slot, a.gen) orelse return null; - if (l.agent_dir.len == 0) return null; - return .{ .name = "agents", .node = activeNode(a), .dir = true, .mode = 0o555 }; - }, - .agent => { - const l = h.live.at(a.slot, a.gen) orelse return null; - var ag: active.Agents = .{}; - active.agentsOf(h.io, l, &ag); - if (a.agent >= ag.count) return null; - const nm = ag.name(a.agent); - @memcpy(h.act_name[0..nm.len], nm); - return .{ .name = h.act_name[0..nm.len], .node = activeNode(a), .dir = true, .mode = 0o555 }; - }, - .transcript, .agent_transcript => { - const path = activePath(h, a) orelse return null; - const st = Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = false }) catch return null; - if (st.kind != .file) return null; - return .{ - .name = a.file.fileName(), - .node = activeNode(a), - .size = st.size, - .mode = 0o444, - .mtime = @truncate(@as(u64, @bitCast(st.mtime.toSeconds()))), - }; - }, - else => { - const text = activeText(h, a) orelse return null; - // Only `zmx` is ever writable, and only when a move is allowed. - const mode: u16 = if (a.file == .zmx and h.allow_move) 0o644 else 0o444; - return .{ .name = a.file.fileName(), .node = activeNode(a), .size = text.len, .mode = mode }; - }, - } -} - -/// The slot holding `pid` for this harness, if any. -fn slotOfPid(h: *Harness, k: active.Kind, pid: u32) ?Act { - for (&h.live.slots, 0..) |*sl, i| { - if (!sl.used or sl.live.kind != k or sl.live.pid != pid) continue; - return .{ .file = .entry, .kind = k, .slot = @intCast(i), .gen = sl.gen }; - } - return null; -} - -fn activeLookup(h: *Harness, req: fs.Req, a: Act, name: []const u8) Answer { - switch (a.file) { - .harness_dir => { - const pid = std.fmt.parseInt(u32, name, 10) catch return fail(req.tag, E.NOENT); - var found = slotOfPid(h, a.kind, pid); - if (found == null) { - rescan(h); - found = slotOfPid(h, a.kind, pid); - } - const act = found orelse return fail(req.tag, E.NOENT); - return activeReply(h, req, act, name); - }, - .entry => { - if (std.mem.eql(u8, name, "agents")) { - return activeReply(h, req, .{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen }, name); - } - for (AFile.entry_files) |f| { - if (!std.mem.eql(u8, f.fileName(), name)) continue; - return activeReply(h, req, .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen }, name); - } - return fail(req.tag, E.NOENT); - }, - .agents => { - const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); - var ag: active.Agents = .{}; - active.agentsOf(h.io, l, &ag); - const i = ag.indexOf(name) orelse return fail(req.tag, E.NOENT); - return activeReply(h, req, .{ - .file = .agent, - .kind = a.kind, - .slot = a.slot, - .gen = a.gen, - .agent = @intCast(i), - }, name); - }, - .agent => { - for (AFile.agent_files) |f| { - if (!std.mem.eql(u8, f.fileName(), name)) continue; - return activeReply(h, req, .{ - .file = f, - .kind = a.kind, - .slot = a.slot, - .gen = a.gen, - .agent = a.agent, - }, name); - } - return fail(req.tag, E.NOENT); - }, - else => return fail(req.tag, E.NOTDIR), - } -} - -fn activeReply(h: *Harness, req: fs.Req, a: Act, name: []const u8) Answer { - const attr = activeAttr(h, a) orelse return fail(req.tag, E.NOENT); - var with_name = attr; - with_name.name = name; - return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; -} - -fn activeReaddir(h: *Harness, req: fs.Req, a: Act) Answer { - var st: Staging = .{ .buf = &h.stage_buf, .skip = req.off }; - switch (a.file) { - .harness_dir => { - for (&h.live.slots, 0..) |*sl, i| { - if (!sl.used or sl.live.kind != a.kind) continue; - var num: [24]u8 = undefined; - const nm = std.fmt.bufPrint(&num, "{d}", .{sl.live.pid}) catch continue; - st.add(activeNode(.{ - .file = .entry, - .kind = a.kind, - .slot = @intCast(i), - .gen = sl.gen, - }), true, nm); - } - }, - .entry => { - const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); - for (AFile.entry_files) |f| { - const child: Act = .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen }; - if (activeAttr(h, child) == null) continue; // no value: not listed - st.add(activeNode(child), false, f.fileName()); - } - if (l.agent_dir.len > 0) { - st.add(activeNode(.{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen }), true, "agents"); - } - }, - .agents => { - const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); - var ag: active.Agents = .{}; - active.agentsOf(h.io, l, &ag); - for (0..ag.count) |i| { - st.add(activeNode(.{ - .file = .agent, - .kind = a.kind, - .slot = a.slot, - .gen = a.gen, - .agent = @intCast(i), - }), true, ag.name(i)); - } - }, - .agent => { - for (AFile.agent_files) |f| { - const child: Act = .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen, .agent = a.agent }; - if (activeAttr(h, child) == null) continue; - st.add(activeNode(child), false, f.fileName()); - } - }, - else => return fail(req.tag, E.NOTDIR), - } - return .{ .reply = .{ .tag = req.tag }, .bytes = h.stage_buf[0..st.len] }; -} - -fn activeRead(h: *Harness, req: fs.Req, a: Act) Answer { - switch (a.file) { - .harness_dir, .entry, .agents, .agent => return fail(req.tag, E.ISDIR), - .transcript, .agent_transcript => { - const path = activePath(h, a) orelse return fail(req.tag, E.NOENT); - const fd = openAbsNoFollow(path) orelse return fail(req.tag, E.NOENT); - const file: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; - defer file.close(h.io); - const st = file.stat(h.io) catch return fail(req.tag, E.IO); - if (st.kind == .directory) return fail(req.tag, E.ISDIR); - if (st.kind != .file) return fail(req.tag, E.PERM); - _ = linux.fcntl(fd, linux.F.SETFL, 0); - const want = @min(req.size, h.data_buf.len); - const n = file.readPositionalAll(h.io, h.data_buf[0..want], req.off) catch - return fail(req.tag, E.IO); - return .{ .reply = .{ .tag = req.tag }, .bytes = h.data_buf[0..n] }; - }, - else => { - const text = activeText(h, a) orelse return fail(req.tag, E.NOENT); - return window(req, text); - }, - } -} - -// ---- the move: writing a zmx session name into an agent's `zmx` ---------------- - -/// How long a move waits, in 100ms ticks: for the agent to take the -/// hangup, then to die outright, then for the new zmx session to post. -const term_ticks: usize = 50; -const kill_ticks: usize = 30; -const post_ticks: usize = 100; - -fn napOneTick(h: *Harness) void { - h.io.sleep(.fromMilliseconds(100), .awake) catch {}; -} - -/// Is this name already a live zmx session? zmx posts each session into -/// the registry, so the registry is the answer — no process scanning. -fn zmxPosted(h: *Harness, name: []const u8) bool { - if (h.runtime_len == 0) return false; - var buf: [active.path_capacity]u8 = undefined; - const path = std.fmt.bufPrint(&buf, "{s}/9p/zmx/{s}", .{ h.runtime_buf[0..h.runtime_len], name }) catch return true; - return Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = true }) != error.FileNotFound; -} - -fn procGone(h: *Harness, pid: u32) bool { - var buf: [active.path_capacity]u8 = undefined; - const path = std.fmt.bufPrint(&buf, "{s}/{d}", .{ h.proc_buf[0..h.proc_len], pid }) catch return false; - _ = Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = true }) catch return true; - return false; -} - -/// SIGTERM, then SIGKILL, then give up. The agent's session was -/// resolved before this ran, so whatever happens it has somewhere to -/// come back to. -fn killAndWait(h: *Harness, pid: u32) bool { - _ = linux.kill(@intCast(pid), .TERM); - for (0..term_ticks) |_| { - if (procGone(h, pid)) return true; - napOneTick(h); - } - _ = linux.kill(@intCast(pid), .KILL); - for (0..kill_ticks) |_| { - if (procGone(h, pid)) return true; - napOneTick(h); - } - return false; -} - -/// `zmx run -d `, in the -/// agent's own directory. Every word but `` is fixed by the -/// harness; `` was checked against zmx's label charset before -/// anything was killed. No byte a client wrote reaches `exec` as a -/// command. -fn spawnZmx(h: *Harness, l: *const active.Live, name: []const u8, resume_value: []const u8) bool { - if (h.zmx_len == 0) return false; - const harness_argv0: []const u8 = @tagName(l.kind); - const resume_flag: []const u8 = switch (l.kind) { - .codex => "resume", - else => "--resume", - }; - - // One buffer holds every NUL-terminated word; `argv` points into it. - var words: [8 * active.path_capacity]u8 = undefined; - var used: usize = 0; - var argv: [9]?[*:0]const u8 = @splat(null); - var n: usize = 0; - const parts = [_][]const u8{ - h.zmx_buf[0..h.zmx_len], "run", name, "-d", - harness_argv0, resume_flag, resume_value, - }; - for (parts) |w| { - if (used + w.len + 1 > words.len or n + 1 >= argv.len) return false; - @memcpy(words[used..][0..w.len], w); - words[used + w.len] = 0; - argv[n] = @ptrCast(&words[used]); - n += 1; - used += w.len + 1; - } - - var cwd_z: [active.cwd_capacity + 1]u8 = @splat(0); - if (l.cwd.len >= cwd_z.len) return false; - @memcpy(cwd_z[0..l.cwd.len], l.cwd.slice()); - - const rc = linux.fork(); - if (linux.errno(rc) != .SUCCESS) return false; - if (rc == 0) { - // The child: only async-signal-safe calls from here. - _ = linux.chdir(@ptrCast(&cwd_z)); - const empty = [_:null]?[*:0]const u8{}; - const envp: cloud9.post.Env = h.envp orelse ∅ - _ = linux.execve(argv[0].?, @ptrCast(&argv), envp); - linux.exit(127); - } - // Reap the forked `zmx`, which returns as soon as the session is up. - const child: i32 = @intCast(rc); - var status: u32 = 0; - for (0..post_ticks) |_| { - const w = linux.waitpid(child, &status, 1); // WNOHANG - if (w == @as(usize, @intCast(child))) break; - napOneTick(h); - } - return true; -} - -/// A move: kill the agent and bring it back inside a zmx session of the -/// name written. Refusals come before anything is destroyed. -fn moveToZmx(h: *Harness, req: fs.Req, a: Act) Answer { - if (!h.allow_move) return fail(req.tag, E.PERM); - const name = std.mem.trim(u8, req.data, " \t\r\n"); - if (!active.legalZmxName(name)) return fail(req.tag, E.INVAL); - - const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); - // Nothing is killed that has nowhere to come back to. - if (l.via == .none or l.session.len == 0) return fail(req.tag, E.PERM); - if (l.cwd.len == 0) return fail(req.tag, E.PERM); - const resume_value: []const u8 = switch (l.kind) { - // omp resumes by the transcript it wrote, the others by id. - .omp => l.transcript.slice(), - .claude, .codex, .hermes => l.session.slice(), - // dsh has no resume form worth guessing at. - .dsh => return fail(req.tag, E.PERM), - }; - if (resume_value.len == 0) return fail(req.tag, E.PERM); - - // Already there: setting a value it already has changes nothing. - var have: [active.text_capacity]u8 = undefined; - if (active.zmxOf(sourcesOf(h), l.pid, &have)) |current| { - if (std.mem.eql(u8, current, name)) { - return .{ .reply = .{ .tag = req.tag, .written = @intCast(req.data.len) } }; - } - } - if (zmxPosted(h, name)) return fail(req.tag, E.EXIST); - // The slot could have gone stale between the scan and this write. - if (!active.stillAlive(sourcesOf(h), l)) return fail(req.tag, E.NOENT); - - // Everything below this line destroys something. - var snapshot = l.*; - if (!killAndWait(h, snapshot.pid)) return fail(req.tag, E.IO); - if (!spawnZmx(h, &snapshot, name, resume_value)) return fail(req.tag, E.IO); - for (0..post_ticks) |_| { - if (zmxPosted(h, name)) { - rescan(h); - return .{ .reply = .{ .tag = req.tag, .written = @intCast(req.data.len) } }; - } - napOneTick(h); - } - rescan(h); - return fail(req.tag, E.IO); -} - -// ---- dispatch ------------------------------------------------------------------ - -/// Answers one engine request. The caller holds `h.mutex` and replies -/// before letting go of it (answer bytes point into `h`). -pub fn handle(h: *Harness, req: fs.Req) Answer { - const t = h.resolve(req.node) orelse return fail(req.tag, E.NOENT); - return switch (req.op) { - .lookup => lookup(h, req, t), - .getattr => attrReply(h, req.tag, t), - .setattr => setattrReq(h, req, t), - .open => open(h, req, t), - .release => release(h, req, t), - .readdir => readdir(h, req, t), - .read => read(h, req, t), - .write => writeReq(h, req, t), - }; -} - -/// Truncation is how a shell's `>` opens a file before writing it. The -/// `zmx` file has no length of its own — a write replaces the value — -/// so on the one writable file a zero-length truncate is a no-op -/// rather than a refusal. Everything else still answers EPERM. -fn setattrReq(h: *Harness, req: fs.Req, t: Target) Answer { - switch (t) { - .act => |a| if (a.file == .zmx and h.allow_move and req.truncate) { - return .{ .reply = .{ .tag = req.tag } }; - }, - else => {}, - } - return fail(req.tag, E.PERM); -} - -/// The tree answers EPERM to every write but one: a zmx session name -/// into a live agent's `zmx`, which moves it there. -fn writeReq(h: *Harness, req: fs.Req, t: Target) Answer { - switch (t) { - .act => |a| if (a.file == .zmx) return moveToZmx(h, req, a), - else => {}, - } - return fail(req.tag, E.PERM); -} - -fn attrReply(h: *Harness, tag: u64, t: Target) Answer { - const a = attrFor(h, t) orelse return fail(tag, E.NOENT); - return .{ .reply = .{ .tag = tag, .attr = a } }; -} - -/// A lookup's attr names the file as the client asked for it (the engine -/// keeps that name for the fid); refs for path targets were already taken -/// by `entryFor`. -fn lookupAttr(h: *Harness, req: fs.Req, t: Target, name: []const u8) Answer { - const a = attrFor(h, t) orelse return fail(req.tag, E.NOENT); - var with_name = a; - with_name.name = name; - return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; -} - -fn lookup(h: *Harness, req: fs.Req, t: Target) Answer { - const name = req.data; - if (name.len == 0 or name.len > 255) return fail(req.tag, E.NOENT); - if (std.mem.indexOfAny(u8, name, "/\x00") != null) return fail(req.tag, E.NOENT); - - if (std.mem.eql(u8, name, ".")) { - // The engine asks for "." when cloning a fid; the reference is - // paid for path targets here like any other lookup result. - switch (t) { - .top, .act => return attrReply(h, req.tag, t), - .path => |e| { - e.refs += 1; - return lookupAttr(h, req, t, name); - }, - } - } - if (std.mem.eql(u8, name, "..")) return lookupParent(h, req, t); - - switch (t) { - .top => |top| switch (top) { - .root => { - const found: ?Top = blk: for (Top.listed) |e| { - if (std.mem.eql(u8, e.fileName(), name)) break :blk e; - } else break :blk null; - return attrReply(h, req.tag, .{ .top = found orelse return fail(req.tag, E.NOENT) }); - }, - .claude, .codex, .skills => { - const m = mountNamed(top, name) orelse return fail(req.tag, E.NOENT); - const e = h.entryFor(m.root, m.rel, true) orelse return fail(req.tag, E.NFILE); - const a = statAttr(h, e) orelse { - unref(e); - return fail(req.tag, E.NOENT); - }; - var with_name = a; - with_name.name = name; - return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; - }, - .omp, .hermes, .dsh => { - const m = top.mirror().?; - return lookupBelow(h, req, m.root, m.rel, name); - }, - .active => { - const k = blk: for (std.enums.values(active.Kind)) |k| { - if (std.mem.eql(u8, k.text(), name)) break :blk k; - } else return fail(req.tag, E.NOENT); - if (!hasLive(h, k)) { - rescan(h); - if (!hasLive(h, k)) return fail(req.tag, E.NOENT); - } - return activeReply(h, req, .{ .file = .harness_dir, .kind = k }, name); - }, - .pid, .uptime => return fail(req.tag, E.NOTDIR), - }, - .path => |e| { - const rel_path = e.rel(); - if (attrFor(h, t)) |a| { - if (!a.dir) return fail(req.tag, E.NOTDIR); - } else return fail(req.tag, E.NOENT); - return lookupBelow(h, req, e.root, rel_path, name); - }, - .act => |a| return activeLookup(h, req, a, name), - } -} - -/// A lookup of `name` in the directory (root, dir_rel): the child's rel -/// path is composed only from the remembered pair and the (single-segment, -/// engine-vetted) name. -fn lookupBelow(h: *Harness, req: fs.Req, r: Root, dir_rel: []const u8, name: []const u8) Answer { - if (excluded(name)) return fail(req.tag, E.NOENT); - var scratch: [rel_capacity + 256]u8 = undefined; - const child_rel = joinRel(&scratch, dir_rel, name) orelse return fail(req.tag, E.NOENT); - const e = h.entryFor(r, child_rel, true) orelse return fail(req.tag, E.NFILE); - const a = statAttr(h, e) orelse { - unref(e); - return fail(req.tag, E.NOENT); - }; - var with_name = a; - with_name.name = name; - return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; -} - -fn joinRel(scratch: []u8, dir_rel: []const u8, name: []const u8) ?[]const u8 { - if (dir_rel.len == 0) { - if (name.len > scratch.len) return null; - @memcpy(scratch[0..name.len], name); - return scratch[0..name.len]; - } - const total = dir_rel.len + 1 + name.len; - if (total > scratch.len) return null; - @memcpy(scratch[0..dir_rel.len], dir_rel); - scratch[dir_rel.len] = '/'; - @memcpy(scratch[dir_rel.len + 1 .. total], name); - return scratch[0..total]; -} - -fn lookupParent(h: *Harness, req: fs.Req, t: Target) Answer { - const parent: Target = switch (t) { - .top => |top| .{ .top = top.parent() }, - .path => |e| blk: { - const rel_path = e.rel(); - if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| { - const up = rel_path[0..i]; - const pe = h.entryFor(e.root, up, true) orelse return fail(req.tag, E.NFILE); - break :blk .{ .path = pe }; - } - break :blk .{ .top = mountOwner(e.root, rel_path) }; - }, - .act => |a| switch (a.file) { - .harness_dir => .{ .top = .active }, - .entry => .{ .act = .{ .file = .harness_dir, .kind = a.kind } }, - .agents => .{ .act = .{ .file = .entry, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, - .agent => .{ .act = .{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, - .agent_model, .agent_transcript => .{ .act = .{ - .file = .agent, - .kind = a.kind, - .slot = a.slot, - .gen = a.gen, - .agent = a.agent, - } }, - // Every other file hangs directly off its entry. - else => .{ .act = .{ .file = .entry, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, - }, - }; - return attrReply(h, req.tag, parent); -} - -fn unref(e: *Entry) void { - if (e.refs > 0) e.refs -= 1; - if (e.refs == 0) { - e.used = false; - e.gen +%= 1; - } -} - -fn open(h: *Harness, req: fs.Req, t: Target) Answer { - _ = attrFor(h, t) orelse return fail(req.tag, E.NOENT); // still there? - // Read-only tree, with exactly one exception: the `zmx` file of a - // live agent, and only when the daemon was started to allow moves. - // Truncation is meaningless there (a write replaces the value), so - // it is accepted rather than refused, which is what `>` needs. - const writable = switch (t) { - .act => |a| a.file == .zmx and h.allow_move, - else => false, - }; - const rw = req.omode & 3; - if (!writable) { - if (rw == cloud9.owrite or rw == cloud9.ordwr) return fail(req.tag, E.PERM); - if (req.omode & cloud9.otrunc != 0) return fail(req.tag, E.PERM); - } - return .{ .reply = .{ .tag = req.tag, .handle = 1 } }; -} - -fn release(h: *Harness, req: fs.Req, t: Target) Answer { - _ = h; - switch (t) { - .top, .act => {}, - .path => |e| unref(e), - } - return .{ .reply = .{ .tag = req.tag } }; -} - -// ---- reads ---------------------------------------------------------------------- - -fn read(h: *Harness, req: fs.Req, t: Target) Answer { - switch (t) { - .top => |top| switch (top) { - .pid, .uptime => { - const text = factText(h, top, &fact_buf); - return window(req, text); - }, - else => return fail(req.tag, E.ISDIR), - }, - .act => |a| return activeRead(h, req, a), - .path => |e| { - // `O_NONBLOCK` so a fifo left in a harness root cannot park the - // daemon in `open`; the kind check below refuses it anyway. - const fd = openIn(h, e.root, e.rel(), .{ - .ACCMODE = .RDONLY, - .NONBLOCK = true, - .CLOEXEC = true, - }) orelse return fail(req.tag, E.NOENT); - const file: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; - defer file.close(h.io); - const st = file.stat(h.io) catch return fail(req.tag, E.IO); - if (st.kind == .directory) return fail(req.tag, E.ISDIR); - // Only regular files have bytes this tree promises to serve. - if (st.kind != .file) return fail(req.tag, E.PERM); - _ = linux.fcntl(fd, linux.F.SETFL, 0); // pread wants no O_NONBLOCK - const want = @min(req.size, h.data_buf.len); - const n = file.readPositionalAll(h.io, h.data_buf[0..want], req.off) catch - return fail(req.tag, E.IO); - return .{ .reply = .{ .tag = req.tag }, .bytes = h.data_buf[0..n] }; - }, - } -} - -/// `text` windowed by the request's offset and size. -fn window(req: fs.Req, text: []const u8) Answer { - const off: usize = @intCast(@min(req.off, text.len)); - const n = @min(text.len - off, req.size); - return .{ .reply = .{ .tag = req.tag }, .bytes = text[off..][0..n] }; -} - -// ---- readdir -------------------------------------------------------------------- - -/// Directory records in the engine's shape: `node:u64le dir:u8 len:u8 name`. -const Staging = struct { - buf: []u8, - len: usize = 0, - skip: u64, - /// Set once a record did not fit. Everything after it is left for the - /// next read: dropping one record and staging a shorter one behind it - /// would lose that entry, because the client's next offset counts the - /// records it received. - full: bool = false, - - fn add(s: *Staging, node: u64, dir: bool, name: []const u8) void { - if (s.full) return; - if (s.skip > 0) { - s.skip -= 1; - return; - } - if (name.len == 0 or name.len > 255) return; - if (s.len + 10 + name.len > s.buf.len) { - s.full = true; - return; - } - std.mem.writeInt(u64, s.buf[s.len..][0..8], node, .little); - s.buf[s.len + 8] = @intFromBool(dir); - s.buf[s.len + 9] = @intCast(name.len); - @memcpy(s.buf[s.len + 10 ..][0..name.len], name); - s.len += 10 + name.len; - } -}; - -fn readdir(h: *Harness, req: fs.Req, t: Target) Answer { - var st: Staging = .{ .buf = &h.stage_buf, .skip = req.off }; - switch (t) { - .top => |top| switch (top) { - .root => for (Top.listed) |e| st.add(topNode(e), e.dir(), e.fileName()), - .claude, .codex, .skills => { - for (named_mounts) |m| { - if (m.owner != top) continue; - // Only mounts whose target exists are listed (a harness - // without skills simply has no skills/ entry). - const a = attrOfMount(h, m.mount.root, m.mount.rel) orelse continue; - const e = h.entryFor(m.mount.root, m.mount.rel, false) orelse continue; - st.add(h.entryNode(e), a.dir, m.name); - } - }, - .pid, .uptime => return fail(req.tag, E.NOTDIR), - .active => { - rescan(h); - for (std.enums.values(active.Kind)) |k| { - if (!hasLive(h, k)) continue; - st.add(activeNode(.{ .file = .harness_dir, .kind = k }), true, k.text()); - } - }, - .omp, .hermes, .dsh => { - const m = top.mirror().?; - if (h.base(m.root) == null) return fail(req.tag, E.NOENT); - switch (listDir(h, &st, m.root, m.rel)) { - .ok => {}, - .gone => return fail(req.tag, E.NOENT), - .failed => return fail(req.tag, E.IO), - .overflow => return fail(req.tag, E.NFILE), - } - }, - }, - .path => |e| { - if (attrFor(h, t)) |a| { - if (!a.dir) return fail(req.tag, E.NOTDIR); - } else return fail(req.tag, E.NOENT); - switch (listDir(h, &st, e.root, e.rel())) { - .ok => {}, - .gone => return fail(req.tag, E.NOENT), - .failed => return fail(req.tag, E.IO), - .overflow => return fail(req.tag, E.NFILE), - } - }, - .act => |a| return activeReaddir(h, req, a), - } - return .{ .reply = .{ .tag = req.tag }, .bytes = h.stage_buf[0..st.len] }; -} - -/// How a listing attempt ended. `overflow` is deliberate: a comptime cap -/// that would silently drop entries answers an error instead, because a -/// short listing is indistinguishable from a small directory. -const Listing = enum { ok, gone, failed, overflow }; - -/// Stages the contents of the directory (root, dir_rel): every name the -/// walker yields that survives the exclusions, sorted so a listing that -/// spans several reads stays consistent. -fn listDir(h: *Harness, st: *Staging, r: Root, dir_rel: []const u8) Listing { - const fd = openIn(h, r, dir_rel, .{ - .ACCMODE = .RDONLY, - .DIRECTORY = true, - .CLOEXEC = true, - }) orelse return .gone; - const dir: Io.Dir = .{ .handle = fd }; - defer Io.Dir.close(dir, h.io); - var read_buf: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; - var reader = Io.Dir.Reader.init(dir, &read_buf); - var names_len: usize = 0; - var count: usize = 0; - while (true) { - const entry = (reader.next(h.io) catch return .failed) orelse break; - if (excluded(entry.name)) continue; - var kind = entry.kind; - if (kind == .unknown or kind == .sym_link) { - // The walker's word is not proof: stat without following. - var child_buf: [rel_capacity + 256]u8 = undefined; - const child_rel = joinRel(&child_buf, dir_rel, entry.name) orelse continue; - const cst = statIn(h, r, child_rel) orelse continue; - kind = cst.kind; - } - if (kind == .sym_link) continue; // never served - // A cap reached is an error, never a short listing: a directory - // that quietly loses entries is a wrong answer, and a caller - // cannot tell it from a small directory. - if (count == list_capacity or names_len + entry.name.len > h.list_names.len) return .overflow; - @memcpy(h.list_names[names_len..][0..entry.name.len], entry.name); - h.list_offs[count] = @intCast(names_len); - h.list_dirs[count] = kind == .directory; - names_len += entry.name.len; - count += 1; - } - // Sort the (offset, length) pairs by name for cursor stability. - const SortCtx = struct { - names: []const u8, - offs: []const u32, - lens: [list_capacity]u32, - - fn lessThan(ctx: @This(), a: usize, b: usize) bool { - return std.mem.order(u8, ctx.nameAt(a), ctx.nameAt(b)) == .lt; - } - fn nameAt(ctx: @This(), i: usize) []const u8 { - const start = ctx.offs[i]; - return ctx.names[start..][0..ctx.lens[i]]; - } - }; - var lens: [list_capacity]u32 = @splat(0); - var order: [list_capacity]usize = @splat(0); - { - var end: usize = 0; - for (0..count) |i| { - end = if (i + 1 < count) h.list_offs[i + 1] else names_len; - lens[i] = @intCast(end - h.list_offs[i]); - order[i] = i; - } - } - const ctx: SortCtx = .{ .names = h.list_names[0..names_len], .offs = &h.list_offs, .lens = lens }; - std.mem.sort(usize, order[0..count], ctx, SortCtx.lessThan); - for (order[0..count]) |i| { - const name = ctx.nameAt(i); - var child_buf: [rel_capacity + 256]u8 = undefined; - const child_rel = joinRel(&child_buf, dir_rel, name) orelse continue; - const e = h.entryFor(r, child_rel, false) orelse return .overflow; // path table full - st.add(h.entryNode(e), h.list_dirs[i], name); - } - return .ok; -} - -// ---- unit tests ------------------------------------------------------------------- - -const testing = std.testing; - -/// A rig with a fake HOME: every root under one temp dir, never the real -/// ~/.claude or any other live harness root. -const Rig = struct { - dir: testing.TmpDir, - /// Build a fixture process tree under /proc and point - /// `/active` at it. Off by default: a unit test must opt in to - /// reading a process tree, and it is never the live one. - with_proc: bool = false, - /// The pid the daemon believes it is, for the ancestry exclusion. - self_pid: u32 = 4242, - zmx: []const u8 = "zmx", - allow_move: bool = false, - path_buf: [std.fs.max_path_bytes]u8 = undefined, - home: []const u8 = undefined, - h: *Harness = undefined, - harness_mem: Harness = undefined, - - fn start(rig: *Rig) !void { - const io = testing.io; - rig.dir = testing.tmpDir(.{}); - errdefer rig.dir.cleanup(); - const len = try rig.dir.dir.realPath(io, &rig.path_buf); - rig.home = try testing.allocator.dupe(u8, rig.path_buf[0..len]); - var mk: [std.fs.max_path_bytes]u8 = undefined; - // The five roots, pinned to the fake home. - inline for ([_][]const u8{ - ".claude/projects/p1", ".claude/skills/revu", - ".codex/sessions/2026/09/21", - ".omp/agent", ".hermes/logs", - ".dsh/profiles", - }) |sub| { - try Io.Dir.cwd().createDirPath(io, try std.fmt.bufPrint(&mk, "{s}/{s}", .{ rig.home, sub })); - } - var base_buf: [5][std.fs.max_path_bytes]u8 = @splat(@splat(0)); - var bases: [5][]const u8 = @splat(""); - inline for (0..5) |i| { - bases[i] = try std.fmt.bufPrint(&base_buf[i], "{s}/{s}", .{ rig.home, home_dirs[i] }); - } - rig.harness_mem = undefined; - var proc_buf: [std.fs.max_path_bytes]u8 = undefined; - const proc_root: []const u8 = if (rig.with_proc) - try std.fmt.bufPrint(&proc_buf, "{s}/proc", .{rig.home}) - else - ""; - if (rig.with_proc) { - try Io.Dir.cwd().createDirPath(io, proc_root); - try rig.put("proc/stat", "cpu 1 2 3\nbtime 1000000\nprocesses 7\n"); - } - rig.harness_mem.init(.{ - .io = io, - .pid = rig.self_pid, - .bases = bases, - .proc = proc_root, - .home = rig.home, - .zmx = rig.zmx, - .allow_move = rig.allow_move, - }); - rig.h = &rig.harness_mem; - } - - fn end(rig: *Rig) void { - rig.dir.cleanup(); - testing.allocator.free(rig.home); - } - - fn put(rig: *Rig, rel_path: []const u8, bytes: []const u8) !void { - var buf: [std.fs.max_path_bytes]u8 = undefined; - const path = try std.fmt.bufPrint(&buf, "{s}/{s}", .{ rig.home, rel_path }); - if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| { - var dbuf: [std.fs.max_path_bytes]u8 = undefined; - const dir_path = try std.fmt.bufPrint(&dbuf, "{s}/{s}", .{ rig.home, rel_path[0..i] }); - try Io.Dir.cwd().createDirPath(testing.io, dir_path); - } - var file = try Io.Dir.createFileAbsolute(testing.io, path, .{}); - defer file.close(testing.io); - try file.writeStreamingAll(testing.io, bytes); - } - - - fn del(rig: *Rig, rel_path: []const u8) !void { - var buf: [std.fs.max_path_bytes]u8 = undefined; - const path = try std.fmt.bufPrint(&buf, "{s}/{s}", .{ rig.home, rel_path }); - try Io.Dir.deleteFileAbsolute(testing.io, path); - } - - /// lookup of `name` in `dir_node`, answering the child's attr. - fn lookupName(rig: *Rig, tag: u64, dir_node: u64, name: []const u8) Answer { - return handle(rig.h, .{ .tag = tag, .op = .lookup, .node = dir_node, .data = name }); - } - - /// Writes one fake process into the fixture `/proc`: the `stat` - /// line (with the start time in field 22, after a name that holds - /// the spaces and parentheses a real one can), the NUL-separated - /// `cmdline`, and a `cwd` symlink. - fn fakeProc(rig: *Rig, o: struct { - pid: u32, - ppid: u32 = 1, - comm: []const u8 = "x", - starttime: u64 = 5000, - argv: []const u8, - cwd: []const u8 = "", - }) !void { - var rel: [128]u8 = undefined; - var stat_text: [512]u8 = undefined; - var w = Io.Writer.fixed(&stat_text); - try w.print("{d} ({s}) S {d}", .{ o.pid, o.comm, o.ppid }); - for (5..22) |i| try w.print(" {d}", .{i}); - try w.print(" {d} 0 0\n", .{o.starttime}); - try rig.put(try std.fmt.bufPrint(&rel, "proc/{d}/stat", .{o.pid}), w.buffered()); - try rig.put(try std.fmt.bufPrint(&rel, "proc/{d}/cmdline", .{o.pid}), o.argv); - - const target = if (o.cwd.len > 0) o.cwd else rig.home; - var link_buf: [std.fs.max_path_bytes]u8 = undefined; - const link = try std.fmt.bufPrintZ(&link_buf, "{s}/proc/{d}/cwd", .{ rig.home, o.pid }); - Io.Dir.cwd().symLink(testing.io, target, link, .{}) catch {}; - } - - /// Removes a fake process, as an exit would. - fn reapProc(rig: *Rig, pid: u32) !void { - var buf: [std.fs.max_path_bytes]u8 = undefined; - const dir_path = try std.fmt.bufPrint(&buf, "{s}/proc/{d}", .{ rig.home, pid }); - const d = try Io.Dir.openDirAbsolute(testing.io, dir_path, .{ .iterate = true }); - var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; - var it = Io.Dir.Reader.init(d, &rb); - var names: [8][64]u8 = undefined; - var lens: [8]usize = @splat(0); - var n: usize = 0; - while (n < names.len) { - const e = (it.next(testing.io) catch break) orelse break; - @memcpy(names[n][0..e.name.len], e.name); - lens[n] = e.name.len; - n += 1; - } - Io.Dir.close(d, testing.io); - for (0..n) |i| { - var one: [std.fs.max_path_bytes]u8 = undefined; - const path = try std.fmt.bufPrint(&one, "{s}/{s}", .{ dir_path, names[i][0..lens[i]] }); - Io.Dir.deleteFileAbsolute(testing.io, path) catch {}; - } - try Io.Dir.cwd().deleteDir(testing.io, dir_path); - } -}; - -const home_dirs = [5][]const u8{ ".claude", ".codex", ".omp", ".hermes", ".dsh" }; - -fn expectNoent(a: Answer) !void { - try testing.expect(a.reply.status == .err); - try testing.expectEqual(E.NOENT, a.reply.errno); -} - -fn expectEperm(a: Answer) !void { - try testing.expect(a.reply.status == .err); - try testing.expectEqual(E.PERM, a.reply.errno); -} - -test "tree: facts at the root" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - const a = rig.lookupName(1, root, "pid"); - try testing.expect(a.reply.status == .ok); - try testing.expect(a.reply.attr.dir == false); - var read_a = handle(rig.h, .{ .tag = 2, .op = .read, .node = a.reply.attr.node, .off = 0, .size = 64 }); - try testing.expectEqualStrings("4242\n", read_a.bytes); - read_a = handle(rig.h, .{ .tag = 3, .op = .read, .node = a.reply.attr.node, .off = 0, .size = 2 }); - try testing.expectEqualStrings("42", read_a.bytes); - const up = rig.lookupName(4, root, "uptime"); - try testing.expect(up.reply.status == .ok); - const up_read = handle(rig.h, .{ .tag = 5, .op = .read, .node = up.reply.attr.node, .off = 0, .size = 64 }); - try testing.expect(up_read.bytes.len > 0 and up_read.bytes[up_read.bytes.len - 1] == '\n'); - // The root lists the facts and the five harness dirs. - const listing = handle(rig.h, .{ .tag = 6, .op = .readdir, .node = root, .off = 0, .size = msize }); - try testing.expect(listing.reply.status == .ok); - for ([_][]const u8{ "pid", "uptime", "claude", "codex", "omp", "hermes", "dsh", "skills" }) |name| { - try testing.expect(stageHas(listing.bytes, name)); - } -} - -fn stageHas(staged: []const u8, name: []const u8) bool { - var i: usize = 0; - while (i + 10 <= staged.len) { - const len: usize = staged[i + 9]; - const entry = staged[i + 10 ..][0..len]; - if (std.mem.eql(u8, entry, name)) return true; - i += 10 + len; - } - return false; -} - -test "tree: exclusions never serve a credentials-shaped name" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - // The fixture: a real transcript and a pile of credentials-shaped names - // at several depths, including inside the mirrored subtrees. - try rig.put(".claude/projects/p1/session-x.jsonl", "{\"a\":1}\n"); - try rig.put(".claude/projects/p1/.credentials.json", "STAY OUT\n"); - try rig.put(".claude/projects/p1/auth.json", "STAY OUT\n"); - try rig.put(".claude/projects/p1/settings.json", "STAY OUT\n"); - try rig.put(".claude/projects/p1/token.txt", "STAY OUT\n"); - try rig.put(".claude/projects/p1/api.key", "STAY OUT\n"); - try rig.put(".claude/projects/p1/models.yml", "STAY OUT\n"); - try rig.put(".claude/projects/p1/deep/.env", "STAY OUT\n"); - try rig.put(".claude/history.jsonl", "{\"h\":1}\n"); - try rig.put(".hermes/auth.json", "STAY OUT\n"); - try rig.put(".hermes/.env", "STAY OUT\n"); - try rig.put(".hermes/config.yaml", "STAY OUT\n"); - try rig.put(".hermes/logs/app.log", "log line\n"); - try rig.put(".dsh/.credentials.yaml", "STAY OUT\n"); - try rig.put(".dsh/settings.yaml", "STAY OUT\n"); - - // /claude lists only its mounts; the credentials at ~/.claude root are - // not in the tree at all (no mount serves them). - const claude = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.claude), .off = 0, .size = msize }); - try testing.expect(claude.reply.status == .ok); - for ([_][]const u8{ "projects", "history", "skills" }) |name| try testing.expect(stageHas(claude.bytes, name)); - try testing.expect(!stageHas(claude.bytes, "credentials")); - - // A project's listing shows the transcript and nothing else. - const projects = rig.lookupName(2, topNode(.claude), "projects"); - try testing.expect(projects.reply.status == .ok); - const p1 = rig.lookupName(3, projects.reply.attr.node, "p1"); - try testing.expect(p1.reply.status == .ok); - const listed = handle(rig.h, .{ .tag = 4, .op = .readdir, .node = p1.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(listed.reply.status == .ok); - try testing.expect(stageHas(listed.bytes, "session-x.jsonl")); - for ([_][]const u8{ ".credentials.json", "auth.json", "settings.json", "token.txt", "api.key", "models.yml" }) |name| { - try testing.expect(!stageHas(listed.bytes, name)); - } - // The plain directory that holds a .env is served; the .env is not. - try testing.expect(stageHas(listed.bytes, "deep")); - - // Every credentials-shaped name is unreachable by lookup too. - for ([_][]const u8{ ".credentials.json", "auth.json", "settings.json", "token.txt", "api.key", "models.yml" }) |name| { - try expectNoent(rig.lookupName(5, p1.reply.attr.node, name)); - } - const deep = rig.lookupName(6, p1.reply.attr.node, "deep"); - try testing.expect(deep.reply.status == .ok); - try expectNoent(rig.lookupName(7, deep.reply.attr.node, ".env")); - - // The fully mirrored roots hide theirs as well. - const hermes = rig.lookupName(8, root, "hermes"); - try testing.expect(hermes.reply.status == .ok); - const hlist = handle(rig.h, .{ .tag = 9, .op = .readdir, .node = hermes.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(stageHas(hlist.bytes, "logs")); - for ([_][]const u8{ "auth.json", ".env", "config.yaml" }) |name| { - try testing.expect(!stageHas(hlist.bytes, name)); - try expectNoent(rig.lookupName(10, hermes.reply.attr.node, name)); - } - const dsh = rig.lookupName(11, root, "dsh"); - try testing.expect(dsh.reply.status == .ok); - const dlist = handle(rig.h, .{ .tag = 12, .op = .readdir, .node = dsh.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(!stageHas(dlist.bytes, ".credentials.yaml")); - try testing.expect(!stageHas(dlist.bytes, "settings.yaml")); - try testing.expect(stageHas(dlist.bytes, "profiles")); - - // The exclusion predicate itself, spelled out. - try testing.expect(excluded(".credentials.json")); - try testing.expect(excluded("AUTH.JSON")); - try testing.expect(excluded("session-token.bin")); - try testing.expect(excluded("id_rsa_backup")); - try testing.expect(excluded("prod.pem")); - try testing.expect(excluded("config.yaml")); - try testing.expect(!excluded("session-x.jsonl")); - try testing.expect(!excluded("SKILL.md")); - try testing.expect(!excluded("logs")); -} - -test "tree: paths compose only from the pinned roots" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); - // A slash, a NUL or an overlong name never reaches the filesystem. - try expectNoent(rig.lookupName(1, topNode(.claude), "projects/../p1")); - try expectNoent(rig.lookupName(2, topNode(.claude), "projects/\x00")); - // A symlink inside a mirror is not served, whatever it points at. - var buf: [std.fs.max_path_bytes]u8 = undefined; - const link = try std.fmt.bufPrintZ(&buf, "{s}/.claude/projects/p1/escape", .{rig.home}); - try Io.Dir.cwd().symLink(testing.io, rig.home, link, .{}); - const projects = rig.lookupName(3, topNode(.claude), "projects"); - const p1 = rig.lookupName(4, projects.reply.attr.node, "p1"); - try expectNoent(rig.lookupName(5, p1.reply.attr.node, "escape")); - // ".." from a mirrored file lands on its canonical parent, and walking - // ".." repeatedly terminates at the root. - const session = rig.lookupName(6, p1.reply.attr.node, "session-x.jsonl"); - try testing.expect(session.reply.status == .ok); - var cur = session.reply.attr.node; - var hops: usize = 0; - while (hops < 8) : (hops += 1) { - const up = rig.lookupName(7, cur, ".."); - try testing.expect(up.reply.status == .ok); - if (up.reply.attr.node == root) break; - cur = up.reply.attr.node; - } - try testing.expect(cur == root or hops < 8); - // The facts refuse lookups with NOTDIR. - try testing.expect(rig.lookupName(8, topNode(.pid), "x").reply.errno == E.NOTDIR); -} - -test "tree: node ids — same file equal, distinct files differ, ids recycle" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); - try rig.put(".claude/projects/p1/session-y.jsonl", "{}\n"); - const projects = rig.lookupName(1, topNode(.claude), "projects"); - const a1 = rig.lookupName(2, projects.reply.attr.node, "p1"); - // Two walks of the same file: the same node id (the dedupe table). - const x1 = rig.lookupName(3, a1.reply.attr.node, "session-x.jsonl"); - const x2 = rig.lookupName(4, a1.reply.attr.node, "session-x.jsonl"); - try testing.expect(x1.reply.status == .ok); - try testing.expectEqual(x1.reply.attr.node, x2.reply.attr.node); - const y1 = rig.lookupName(5, a1.reply.attr.node, "session-y.jsonl"); - try testing.expect(y1.reply.attr.node != x1.reply.attr.node); - // The union skills view and the harness's own skills share identity. - const via_harness = rig.lookupName(6, topNode(.claude), "skills"); - const via_union = rig.lookupName(7, topNode(.skills), "claude"); - try testing.expectEqual(via_harness.reply.attr.node, via_union.reply.attr.node); - // References are paid back by release: both slots go, ids recycle. - for ([_]u64{ x1.reply.attr.node, x2.reply.attr.node, y1.reply.attr.node, a1.reply.attr.node }) |n| { - const rel = handle(rig.h, .{ .tag = 8, .op = .release, .node = n }); - try testing.expect(rel.reply.status == .ok); - } - const x3 = rig.lookupName(9, projects.reply.attr.node, "p1"); - const x4 = rig.lookupName(10, x3.reply.attr.node, "session-x.jsonl"); - // The entry was freed and re-created: a fresh generation, a new id. - try testing.expect(x4.reply.attr.node != x1.reply.attr.node); - // Node packing round-trips through the struct. - const n: Node = .{ .idx = 3, .kind = 1, .serial = 0x1234_5678_9abc }; - const bits: u64 = @bitCast(n); - const back: Node = @bitCast(bits); - try testing.expect(back.idx == 3 and back.kind == 1 and back.serial == 0x1234_5678_9abc); -} - -test "tree: a file that vanishes between lookup and read answers ENOENT" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put(".claude/projects/p1/session-x.jsonl", "{\"a\":1}\n"); - const projects = rig.lookupName(1, topNode(.claude), "projects"); - const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); - const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); - try testing.expect(session.reply.status == .ok); - // The transcript reads back whole, fresh from disk. - const whole = handle(rig.h, .{ .tag = 4, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 1024 }); - try testing.expectEqualStrings("{\"a\":1}\n", whole.bytes); - // It vanishes: lookup, getattr and read all answer ENOENT cleanly. - try rig.del(".claude/projects/p1/session-x.jsonl"); - try expectNoent(rig.lookupName(5, p1.reply.attr.node, "session-x.jsonl")); - const gone = handle(rig.h, .{ .tag = 6, .op = .getattr, .node = session.reply.attr.node }); - try expectNoent(gone); - const read_gone = handle(rig.h, .{ .tag = 7, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 64 }); - try expectNoent(read_gone); -} - -test "tree: writes and setattrs answer EPERM" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); - const projects = rig.lookupName(1, topNode(.claude), "projects"); - const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); - const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); - try testing.expect(session.reply.status == .ok); - const write = handle(rig.h, .{ .tag = 4, .op = .write, .node = session.reply.attr.node, .data = "x" }); - try expectEperm(write); - const set = handle(rig.h, .{ .tag = 5, .op = .setattr, .node = session.reply.attr.node, .set = .{ .mtime = true }, .mtime = 1 }); - try expectEperm(set); - // An open for write is refused at open time. - const wopen = handle(rig.h, .{ .tag = 6, .op = .open, .node = session.reply.attr.node, .omode = cloud9.owrite }); - try expectEperm(wopen); - const ropen = handle(rig.h, .{ .tag = 7, .op = .open, .node = session.reply.attr.node, .omode = cloud9.oread }); - try testing.expect(ropen.reply.status == .ok); -} - -test "tree: a transcript written while serving is visible at once" { - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put(".claude/projects/p1/session-x.jsonl", "{\"n\":1}\n"); - const projects = rig.lookupName(1, topNode(.claude), "projects"); - const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); - // The harness appends; the next read sees it — no cache in between. - try rig.put(".claude/projects/p1/session-x.jsonl", "{\"n\":1}\n{\"n\":2}\n"); - const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); - const fresh = handle(rig.h, .{ .tag = 4, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 1024 }); - try testing.expectEqualStrings("{\"n\":1}\n{\"n\":2}\n", fresh.bytes); - // A brand-new session file appears on the next listing. - try rig.put(".claude/projects/p1/session-new.jsonl", "{\"n\":9}\n"); - const listed = handle(rig.h, .{ .tag = 5, .op = .readdir, .node = p1.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(stageHas(listed.bytes, "session-new.jsonl")); -} - -comptime { - // Every static mount target must survive the exclusion rules; the - // daemon's own skeleton may never be filtered out from under it. - for (named_mounts) |m| { - if (excludedPath(m.mount.rel)) @compileError("a 9harness mount target is excluded by name"); - } -} - -fn stageCount(staged: []const u8) usize { - var i: usize = 0; - var n: usize = 0; - while (i + 10 <= staged.len) : (n += 1) i += 10 + @as(usize, staged[i + 9]); - return n; -} - -test "tree: a file directly under a mirror root reads back" { - // Regression: joining a child onto an empty relative path returned an - // uncopied scratch slice, so every file at the top of a mirror root - // (/hermes/, /dsh/) listed but resolved to garbage bytes. - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put(".hermes/note.txt", "hello-hermes\n"); - const listing = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.hermes), .off = 0, .size = msize }); - try testing.expect(stageHas(listing.bytes, "note.txt")); - const note = rig.lookupName(2, topNode(.hermes), "note.txt"); - try testing.expect(note.reply.status == .ok); - const bytes = handle(rig.h, .{ .tag = 3, .op = .read, .node = note.reply.attr.node, .off = 0, .size = 64 }); - try testing.expectEqualStrings("hello-hermes\n", bytes.bytes); -} - -test "tree: a name swapped for a symlink under an open handle serves nothing" { - // Regression: the read path composed the whole path and opened it in - // one call, following symlinks. A name replaced between the walk and - // the read handed the client bytes from outside every pinned root. - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - try rig.put("outside.txt", "OUTSIDE-THE-ROOTS\n"); // beside the roots, not in one - try rig.put(".claude/projects/p1/swap.txt", "safe\n"); - const projects = rig.lookupName(1, topNode(.claude), "projects"); - const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); - const swap = rig.lookupName(3, p1.reply.attr.node, "swap.txt"); - try testing.expect(swap.reply.status == .ok); - const opened = handle(rig.h, .{ .tag = 4, .op = .open, .node = swap.reply.attr.node, .omode = cloud9.oread }); - try testing.expect(opened.reply.status == .ok); - // The file becomes a symlink out of the tree while the handle is open. - var link_buf: [std.fs.max_path_bytes]u8 = undefined; - var target_buf: [std.fs.max_path_bytes]u8 = undefined; - const link = try std.fmt.bufPrintZ(&link_buf, "{s}/.claude/projects/p1/swap.txt", .{rig.home}); - const target = try std.fmt.bufPrint(&target_buf, "{s}/outside.txt", .{rig.home}); - try rig.del(".claude/projects/p1/swap.txt"); - try Io.Dir.cwd().symLink(testing.io, target, link, .{}); - const after = handle(rig.h, .{ .tag = 5, .op = .read, .node = swap.reply.attr.node, .off = 0, .size = 64 }); - try expectNoent(after); - try testing.expectEqual(@as(usize, 0), after.bytes.len); -} - -test "tree: a listing past the cap fails loudly instead of truncating" { - // Regression: a directory with more entries (or longer names) than the - // comptime caps was served short, and a short listing is - // indistinguishable from a small directory. - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - var name_buf: [64]u8 = undefined; - for (0..list_capacity + 1) |i| { - try rig.put(try std.fmt.bufPrint(&name_buf, ".dsh/f{d:0>5}", .{i}), ""); - } - const listed = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.dsh), .off = 0, .size = msize }); - try testing.expect(listed.reply.status == .err); - try testing.expectEqual(E.NFILE, listed.reply.errno); -} - -test "tree: a listing spanning several reads loses no entry" { - // The staging buffer fills long before a big directory ends. Staging - // must stop at the first record that does not fit: dropping it and - // packing a shorter one behind it would lose that entry, because the - // next read's offset counts the records already delivered. - var rig: Rig = .{ .dir = undefined }; - try rig.start(); - defer rig.end(); - const count = 200; - var name_buf: [320]u8 = undefined; - for (0..count) |i| { - const name = try std.fmt.bufPrint(&name_buf, ".dsh/{d:0>3}{s}", .{ i, "n" ** 200 }); - try rig.put(name, ""); - } - var seen: [count]bool = @splat(false); - var off: u64 = 0; - var reads: usize = 0; - while (reads < 16) : (reads += 1) { - const page = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.dsh), .off = off, .size = msize }); - try testing.expect(page.reply.status == .ok); - if (page.bytes.len == 0) break; - var i: usize = 0; - while (i + 10 <= page.bytes.len) { - const len: usize = page.bytes[i + 9]; - const name = page.bytes[i + 10 ..][0..len]; - i += 10 + len; - // The fixture root also holds `profiles`, created by the rig. - const idx = std.fmt.parseInt(usize, name[0..@min(3, name.len)], 10) catch continue; - try testing.expect(!seen[idx]); // never served twice - seen[idx] = true; - } - off += stageCount(page.bytes); - } - try testing.expect(reads > 1); // the listing really did span several reads - for (seen) |s| try testing.expect(s); -} - -test { - _ = active; // the derived view's own tests run with the tree's -} - -// ---- /active: the derived view ------------------------------------------------ - -/// Walks `/active` down to one agent's directory, answering its node. -fn activeEntryNode(rig: *Rig, harness: []const u8, pid: []const u8) !u64 { - const act = rig.lookupName(90, root, "active"); - try testing.expect(act.reply.status == .ok); - // A listing is what rescans, so it comes before the walk. - _ = handle(rig.h, .{ .tag = 91, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); - const h_dir = rig.lookupName(92, act.reply.attr.node, harness); - try testing.expect(h_dir.reply.status == .ok); - const entry = rig.lookupName(93, h_dir.reply.attr.node, pid); - try testing.expect(entry.reply.status == .ok); - return entry.reply.attr.node; -} - -fn activeField(rig: *Rig, entry: u64, name: []const u8) ![]const u8 { - const f = rig.lookupName(94, entry, name); - try testing.expect(f.reply.status == .ok); - const r = handle(rig.h, .{ .tag = 95, .op = .read, .node = f.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(r.reply.status == .ok); - return r.bytes; -} - -test "active: a fixture process tree lists by harness and by pid" { - var rig: Rig = .{ .dir = undefined, .with_proc = true }; - try rig.start(); - defer rig.end(); - - // A Claude Code session, resolved the way the harness publishes it. - try rig.fakeProc(.{ .pid = 1001, .comm = "claude", .starttime = 5000, .argv = "claude\x00--print\x00" }); - var rec: [512]u8 = undefined; - try rig.put(".claude/sessions/1001.json", try std.fmt.bufPrint(&rec, - \\{{"pid":1001,"sessionId":"sess-abc","cwd":"{s}", - \\ "name":"fixture-one","status":"busy","procStart":"5000"}} - , .{rig.home})); - var slug_buf: [512]u8 = undefined; - const slug = active.slugOf(.claude, rig.home, rig.home, &slug_buf).?; - var tr: [640]u8 = undefined; - try rig.put( - try std.fmt.bufPrint(&tr, ".claude/projects/{s}/sess-abc.jsonl", .{slug}), - "{\"type\":\"summary\",\"aiTitle\":\"Fixture Title\"}\n", - ); - - const entry = try activeEntryNode(&rig, "claude", "1001"); - try testing.expectEqualStrings("1001\n", try activeField(&rig, entry, "pid")); - try testing.expectEqualStrings("sess-abc\n", try activeField(&rig, entry, "session")); - try testing.expectEqualStrings("registry\n", try activeField(&rig, entry, "via")); - try testing.expectEqualStrings("fixture-one\n", try activeField(&rig, entry, "name")); - try testing.expectEqualStrings("busy\n", try activeField(&rig, entry, "status")); - try testing.expectEqualStrings("Fixture Title\n", try activeField(&rig, entry, "title")); - // started is the boot time plus the process's own, in seconds. - try testing.expectEqualStrings("1000050\n", try activeField(&rig, entry, "started")); - // The transcript is served as the file it is, not as a field. - const t = rig.lookupName(96, entry, "transcript"); - try testing.expect(t.reply.status == .ok); - const bytes = handle(rig.h, .{ .tag = 97, .op = .read, .node = t.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(std.mem.indexOf(u8, bytes.bytes, "Fixture Title") != null); - // A field this harness does not publish is absent, not blank. - try expectNoent(rig.lookupName(98, entry, "model")); -} - -test "active: a daemon or helper is never listed as an agent" { - var rig: Rig = .{ .dir = undefined, .with_proc = true }; - try rig.start(); - defer rig.end(); - // The shapes actually running on this machine. - try rig.fakeProc(.{ .pid = 1010, .comm = "python3", .argv = "python3\x00-m\x00hermes_cli.main\x00gateway\x00run\x00" }); - try rig.fakeProc(.{ .pid = 1011, .comm = "omp", .argv = "omp\x00__omp_worker_daemon_broker\x00" }); - try rig.fakeProc(.{ .pid = 1012, .comm = "claude", .argv = "claude\x00mcp-server\x00" }); - // ...and one real session, so an empty answer cannot pass by default. - try rig.fakeProc(.{ .pid = 1013, .comm = "claude", .argv = "claude\x00" }); - - const act = rig.lookupName(1, root, "active"); - const listing = handle(rig.h, .{ .tag = 2, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(stageHas(listing.bytes, "claude")); - try testing.expect(!stageHas(listing.bytes, "hermes")); - try testing.expect(!stageHas(listing.bytes, "omp")); - const claude = rig.lookupName(3, act.reply.attr.node, "claude"); - const pids = handle(rig.h, .{ .tag = 4, .op = .readdir, .node = claude.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(stageHas(pids.bytes, "1013")); - try testing.expect(!stageHas(pids.bytes, "1012")); // the mcp server -} - -test "active: a pid reused by another process stops resolving" { - var rig: Rig = .{ .dir = undefined, .with_proc = true }; - try rig.start(); - defer rig.end(); - try rig.fakeProc(.{ .pid = 1020, .comm = "claude", .starttime = 5000, .argv = "claude\x00" }); - const entry = try activeEntryNode(&rig, "claude", "1020"); - try testing.expect(handle(rig.h, .{ .tag = 5, .op = .getattr, .node = entry }).reply.status == .ok); - - // The same pid, a different process: only the start time says so. - try rig.fakeProc(.{ .pid = 1020, .comm = "claude", .starttime = 9999, .argv = "claude\x00" }); - const act = rig.lookupName(6, root, "active"); - _ = handle(rig.h, .{ .tag = 7, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); - try expectNoent(handle(rig.h, .{ .tag = 8, .op = .getattr, .node = entry })); - // The pid is still there — as the new process, under a new node. - const fresh = try activeEntryNode(&rig, "claude", "1020"); - try testing.expect(fresh != entry); -} - -test "active: a process that exits leaves the tree" { - var rig: Rig = .{ .dir = undefined, .with_proc = true }; - try rig.start(); - defer rig.end(); - try rig.fakeProc(.{ .pid = 1030, .comm = "claude", .argv = "claude\x00" }); - const entry = try activeEntryNode(&rig, "claude", "1030"); - try testing.expect(handle(rig.h, .{ .tag = 9, .op = .getattr, .node = entry }).reply.status == .ok); - - try rig.reapProc(1030); - const act = rig.lookupName(10, root, "active"); - const listing = handle(rig.h, .{ .tag = 11, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(!stageHas(listing.bytes, "claude")); - try expectNoent(handle(rig.h, .{ .tag = 12, .op = .getattr, .node = entry })); - try expectNoent(rig.lookupName(13, act.reply.attr.node, "claude")); -} - -test "active: the daemon never lists the process tree it lives in" { - // 1041 is the daemon, its parent 1040 is a harness: showing it would - // let a client act on the tree serving it. - var rig: Rig = .{ .dir = undefined, .with_proc = true, .self_pid = 1041 }; - try rig.start(); - defer rig.end(); - try rig.fakeProc(.{ .pid = 1040, .comm = "claude", .argv = "claude\x00" }); - try rig.fakeProc(.{ .pid = 1041, .comm = "9harness", .ppid = 1040, .argv = "9harness\x00" }); - try rig.fakeProc(.{ .pid = 1042, .comm = "claude", .argv = "claude\x00" }); - - const act = rig.lookupName(14, root, "active"); - const claude = rig.lookupName(15, act.reply.attr.node, "claude"); - try testing.expect(claude.reply.status == .ok); - const pids = handle(rig.h, .{ .tag = 16, .op = .readdir, .node = claude.reply.attr.node, .off = 0, .size = msize }); - try testing.expect(stageHas(pids.bytes, "1042")); // an unrelated session lists - try testing.expect(!stageHas(pids.bytes, "1040")); // its own parent does not -} diff --git a/9harness/test/e2e.sh b/9harness/test/e2e.sh deleted file mode 100755 index 254d999..0000000 --- a/9harness/test/e2e.sh +++ /dev/null @@ -1,347 +0,0 @@ -#!/usr/bin/env bash -# End-to-end suite for 9harness: the read-only, fresh-from-disk 9P view of -# every harness's state. -# -# Usage: bash 9harness/test/e2e.sh <9harness> [<9ns>] (zig build 9harness-itest) -# -# Part A always runs, entirely on fixtures: a fake home with fixture roots -# pinned by --root, a scratch XDG_RUNTIME_DIR registry, plan9port's 9p as -# the client. It proves: the posted name is listed, the roots walk and -# read, a transcript comes back byte-identical (cmp), credentials-shaped -# files are unreachable, a file appended after the daemon started is -# visible at once, writes answer EPERM, and --unix/--fd listen forms work. -# -# Part B (the money shot) runs against the REAL roots and the REAL -# registry only when the `harness` name is free, and only reads: the posted -# name in /run/user//9p, 9p walks of the live ~/.claude, a real -# transcript byte-identical through the tree, the 9ns --mntgen mount of -# /mnt/9p/harness, and the live ~/.claude/.credentials.json unreachable. -# It is skipped (not failed) when a live daemon already owns the name. -# -# Exit 0 on success (or when the machine cannot run a part), 1 on failure. -set -u - -H9=$(realpath "${1:?path to 9harness}") -HARNESS_TMP_XDG="${XDG_RUNTIME_DIR:-}" -[ $# -ge 2 ] && [ -n "$2" ] && NS=$(realpath "$2") -P9P=/usr/lib/plan9/bin/9p -TMP=$(mktemp -d "${TMPDIR:-/tmp}/9harness.XXXXXX") -PIDS=() -FAILED=0 -PASSED=0 - -cleanup() { - for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done - rm -rf "$TMP" -} -trap cleanup EXIT - -pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } -fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } -expect_eq() { # name expected actual - if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi -} -expect_contains() { # name needle haystack - case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac -} -expect_missing() { # name needle haystack - case "$3" in *"$2"*) fail "$1" "found: $(printf %q "$2")" "in: $(printf %q "$3")" ;; *) pass "$1" ;; esac -} - -if [ ! -x "$P9P" ]; then - echo "SKIP: plan9port 9p not at $P9P"; exit 0 -fi -if ! unshare -Urm true 2>/dev/null; then - echo "SKIP: unprivileged user namespaces unavailable"; exit 0 -fi - -start_daemon() { # args... -> sets DAEMON_PID, logs to $TMP/daemon.log - "$H9" "$@" >"$TMP/daemon.log" 2>&1 & - DAEMON_PID=$! - PIDS+=("$DAEMON_PID") -} -wait_posted() { # socket path - for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done - return 1 -} -p9() { "$P9P" -a "unix!$1" "${@:2}"; } - -# ============================================================================ -echo "# part A: fixture roots, scratch registry" -# ============================================================================ -export XDG_RUNTIME_DIR="$TMP" -REG="$TMP/9p" -FX="$TMP/home" -mkdir -p "$FX" - -# The fixture home: five roots, a real-shaped claude project with a -# transcript, credentials-shaped files at several depths, codex sessions, -# an omp agent dir, hermes logs and dsh profiles. -mkfile() { mkdir -p "$(dirname "$1")"; printf '%s' "$2" > "$1"; } -mkfile "$FX/claude/projects/-tmp-proj/session-abc.jsonl" '{"type":"user","message":"first line"} -{"type":"assistant","message":"second line"} -' -mkfile "$FX/claude/projects/-tmp-proj/.credentials.json" 'STAY OUT' -mkfile "$FX/claude/projects/-tmp-proj/auth.json" 'STAY OUT' -mkfile "$FX/claude/projects/-tmp-proj/token.bin" 'STAY OUT' -mkfile "$FX/claude/history.jsonl" '{"display":"claude prompt one"} -{"display":"claude prompt two"} -' -mkdir -p "$FX/claude/skills/revu" -mkfile "$FX/claude/skills/revu/SKILL.md" '# revu skill' -mkfile "$FX/codex/sessions/2026/09/21/rollout-x.jsonl" '{"session":"codex one"} -' -mkfile "$FX/codex/session_index.jsonl" '{"id":"x"} -' -mkfile "$FX/codex/history.jsonl" '{"text":"codex prompt"} -' -mkfile "$FX/omp/agent/history.db" 'fake-history-db' -mkfile "$FX/omp/agent/config.yml" 'STAY OUT' -mkfile "$FX/hermes/auth.json" 'STAY OUT' -mkfile "$FX/hermes/logs/app.log" 'hermes log line -' -mkfile "$FX/hermes/state.db" 'fake-hermes-state' -mkfile "$FX/dsh/profiles/p1.yaml" 'name: p1' -mkfile "$FX/dsh/.credentials.yaml" 'STAY OUT' - -start_daemon --root "claude=$FX/claude" --root "codex=$FX/codex" \ - --root "omp=$FX/omp" --root "hermes=$FX/hermes" --root "dsh=$FX/dsh" \ - --name harness -wait_posted "$REG/harness" || { fail "the daemon posts as harness" "$(cat "$TMP/daemon.log")"; exit 1; } - -# 1. The posted name is listed. -expect_contains "posted name listed in the registry" harness "$(ls "$REG")" - -# 2. The roots walk; a real transcript reads back byte-identical. -expect_contains "ls / shows the roots" claude "$(p9 "$REG/harness" ls /)" -expect_contains "ls / shows codex" codex "$(p9 "$REG/harness" ls /)" -expect_contains "ls / shows skills" skills "$(p9 "$REG/harness" ls /)" -p9 "$REG/harness" read /claude/projects/-tmp-proj/session-abc.jsonl > "$TMP/via-9p.jsonl" 2>"$TMP/read.err" \ - || fail "transcript read through the tree" "$(cat "$TMP/read.err")" -cmp -s "$TMP/via-9p.jsonl" "$FX/claude/projects/-tmp-proj/session-abc.jsonl" \ - && pass "transcript byte-identical through the tree (cmp)" \ - || fail "transcript byte-identical through the tree (cmp)" "differs" -expect_eq "history file reads" "$(cat "$FX/claude/history.jsonl")" "$(p9 "$REG/harness" read /claude/history)" -expect_eq "skills union mirrors the harness tree" "$(cat "$FX/claude/skills/revu/SKILL.md")" \ - "$(p9 "$REG/harness" read /skills/claude/revu/SKILL.md)" - -# 3. A credentials-shaped file is unreachable anywhere in the tree. -PROJ_LS=$(p9 "$REG/harness" ls /claude/projects/-tmp-proj) -for bad in .credentials.json auth.json token.bin; do - expect_missing "credentials-shaped $bad not listed" "$bad" "$PROJ_LS" - p9 "$REG/harness" read "/claude/projects/-tmp-proj/$bad" >/dev/null 2>&1 \ - && fail "credentials-shaped $bad unreachable" "read succeeded" \ - || pass "credentials-shaped $bad unreachable" -done -expect_missing "hermes auth.json not listed" auth.json "$(p9 "$REG/harness" ls /hermes)" -expect_missing "omp config.yml not listed" config.yml "$(p9 "$REG/harness" ls /omp)" -expect_missing "dsh .credentials.yaml not listed" credentials.yaml "$(p9 "$REG/harness" ls /dsh)" -sleep 0.3 -expect_contains "hermes logs still served" logs "$(p9 "$REG/harness" ls /hermes)" - -# 3b. A file at the very top of a mirror root: its relative path is the -# bare name, the one case a join onto an empty directory path gets wrong. -expect_contains "a file at the top of a mirror root is listed" state.db "$(p9 "$REG/harness" ls /hermes)" -expect_eq "a file at the top of a mirror root reads back" "$(cat "$FX/hermes/state.db")" \ - "$(p9 "$REG/harness" read /hermes/state.db)" - -# 3c. A directory bigger than the listing caps fails loudly. A short -# listing is indistinguishable from a small directory, so the daemon must -# never answer one: the read errors and the client sees it. -mkdir -p "$FX/dsh/wide" -seq 1 1100 | while read -r i; do : > "$FX/dsh/wide/f$(printf %05d "$i")"; done -if p9 "$REG/harness" ls /dsh/wide > "$TMP/wide.out" 2>"$TMP/wide.err"; then - fail "an over-cap directory fails instead of truncating" "listed $(wc -l < "$TMP/wide.out") entries" -else - pass "an over-cap directory fails instead of truncating ($(head -c 80 "$TMP/wide.err"))" -fi -rm -rf "$FX/dsh/wide" - -# 4. Live visibility: a file appended after the daemon started. -printf '{"type":"assistant","message":"third line"}\n' >> "$FX/claude/projects/-tmp-proj/session-abc.jsonl" -expect_eq "append after start is visible immediately" \ - "$(cat "$FX/claude/projects/-tmp-proj/session-abc.jsonl")" \ - "$(p9 "$REG/harness" read /claude/projects/-tmp-proj/session-abc.jsonl)" -mkdir -p "$FX/claude/projects/-tmp-proj2" -mkfile "$FX/claude/projects/-tmp-proj2/session-new.jsonl" '{"new":true} -' -expect_contains "new session dir appears at once" -tmp-proj2 "$(p9 "$REG/harness" ls /claude/projects)" - -# 5. The facts and the write refusal. -DAEMON_PID_TEXT=$(p9 "$REG/harness" read /pid) -expect_eq "pid fact answers the daemon's pid" "$DAEMON_PID" "$DAEMON_PID_TEXT" -[ -n "$(p9 "$REG/harness" read /uptime)" ] && pass "uptime fact answers" || fail "uptime fact answers" "empty" -printf 'x' | p9 "$REG/harness" write /pid >/dev/null 2>&1 \ - && fail "write answers EPERM" "write succeeded" \ - || pass "write answers EPERM" - -# 6. The --unix listen form beside the post. -"$H9" --unix "$TMP/plain.sock" --no-post --root "claude=$FX/claude" >"$TMP/d2.log" 2>&1 & -PIDS+=($!) -for _ in $(seq 1 100); do [ -S "$TMP/plain.sock" ] && break; sleep 0.05; done -expect_eq "--unix listen form serves" "$(cat "$FX/claude/history.jsonl")" "$(p9 "$TMP/plain.sock" read /claude/history)" - -# 7. The --fd listen form: one 9P session over a connected stream fd -# (the 9ns --spawn / socket-activation shape), driven through a -# socketpair: the daemon sees EOF when both ends close and exits. -if command -v python3 >/dev/null; then - python3 - "$H9" "$FX" <<'EOF' -import os, socket, subprocess, sys -h9, fx = sys.argv[1], sys.argv[2] -a, b = socket.socketpair() -pid = os.fork() -if pid == 0: - a.close() - os.dup2(b.fileno(), 7) - os.execv(h9, [h9, "--fd", "7", "--root", f"claude={fx}/claude"]) -b.close() -a.close() -os.waitpid(pid, 0) -EOF - pass "--fd listen form runs a session and exits on hangup" -else - echo "skip - --fd form: python3 not available" -fi - -kill "$DAEMON_PID" 2>/dev/null -wait "$DAEMON_PID" 2>/dev/null -sleep 0.2 -[ -S "$REG/harness" ] && fail "SIGTERM unposts the name" "socket still there" || pass "SIGTERM unposts the name" - -# ============================================================================ -echo "# part C: /active, the derived view" -# ============================================================================ -# A fake agent: a process whose argv[0] basename is a harness name, with a -# session record Claude Code's own shape. The proc root is the real /proc -# (the fixture seam is unit-tested); nothing real is ever killed, because -# the only process this part touches is the sleep it started itself. -ACT="$TMP/act" -mkdir -p "$ACT/bin" "$ACT/home/.claude/sessions" "$ACT/home/.claude/projects" "$ACT/rt/9p/zmx" -cp /bin/sleep "$ACT/bin/claude" -# A zmx that records how it was called and posts the name, as the real one -# does; a move must never be tested against the user's live sessions. -cat > "$ACT/bin/zmx" < "$ACT/zmx-argv" -: > "$ACT/rt/9p/zmx/\$2" -exit 0 -ZEOF -chmod +x "$ACT/bin/zmx" - -"$ACT/bin/claude" 600 & -AGENT=$! -PIDS+=("$AGENT") -sleep 0.3 -# Field 22 of /proc//stat, counted from the last ')' — the executable -# name can hold spaces and parentheses, so the line is never just split. -AGENT_START=$(sed 's/.*) //' "/proc/$AGENT/stat" | awk '{print $20}') -AGENT_CWD=$(readlink "/proc/$AGENT/cwd") -printf '{"pid":%d,"sessionId":"sess-e2e","cwd":"%s","name":"e2e-agent","status":"idle","procStart":"%s"}\n' \ - "$AGENT" "$AGENT_CWD" "$AGENT_START" > "$ACT/home/.claude/sessions/$AGENT.json" - -act_roots() { - echo --root "claude=$ACT/home/.claude" --root "codex=$ACT/home/.codex" \ - --root "omp=$ACT/home/.omp" --root "hermes=$ACT/home/.hermes" --root "dsh=$ACT/home/.dsh" -} - -# First: the read-only default. No --allow-move, so zmx cannot be written. -XDG_RUNTIME_DIR="$ACT/rt" "$H9" --no-post --unix "$ACT/ro.sock" $(act_roots) >"$ACT/ro.log" 2>&1 & -PIDS+=("$!") -wait_posted "$ACT/ro.sock" || { fail "read-only daemon listens" "$(cat "$ACT/ro.log")"; } -expect_contains "the agent lists under its harness" "$AGENT" "$(p9 "$ACT/ro.sock" ls /active/claude)" -expect_eq "its session comes from the harness's own record" "sess-e2e" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/session")" -expect_eq "the route taken is named" "registry" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/via")" -expect_eq "the harness's own name is served" "e2e-agent" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/name")" -expect_eq "the harness's own status is served" "idle" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/status")" -if echo "nope" | p9 "$ACT/ro.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null; then - fail "without --allow-move the tree stays read-only" "the write was accepted" -else - pass "without --allow-move the tree stays read-only" -fi -expect_eq "a refused move leaves the agent running" "yes" "$([ -d "/proc/$AGENT" ] && echo yes)" - -# Now a daemon that allows moves. -XDG_RUNTIME_DIR="$ACT/rt" "$H9" --no-post --unix "$ACT/rw.sock" $(act_roots) \ - --allow-move --zmx "$ACT/bin/zmx" >"$ACT/rw.log" 2>&1 & -PIDS+=("$!") -wait_posted "$ACT/rw.sock" || { fail "move-allowing daemon listens" "$(cat "$ACT/rw.log")"; } - -# Every refusal must come before anything is destroyed. -for bad in "has space" "../escape" "semi;colon"; do - echo "$bad" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null - if [ $? -eq 0 ]; then fail "an illegal zmx name is refused ($bad)"; else pass "an illegal zmx name is refused ($bad)"; fi -done -: > "$ACT/rt/9p/zmx/occupied" -echo "occupied" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null \ - && fail "a name a live session holds is refused" || pass "a name a live session holds is refused" -expect_eq "no refusal killed the agent" "yes" "$([ -d "/proc/$AGENT" ] && echo yes)" - -# The move itself. -if echo "e2e-moved" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>"$ACT/move.err"; then - pass "a legal name moves the agent" -else - fail "a legal name moves the agent" "$(cat "$ACT/move.err")" -fi -sleep 0.4 -expect_eq "the agent it replaced is gone" "gone" "$([ -d "/proc/$AGENT" ] || echo gone)" -# The command is fixed by the harness: the client supplied only the name. -expect_eq "zmx ran the harness with its session resumed" \ - "run e2e-moved -d claude --resume sess-e2e" "$(cat "$ACT/zmx-argv")" -expect_missing "no client byte reached the command" "e2e-moved -d claude --resume sess-e2e;" "$(cat "$ACT/zmx-argv")" - -# ============================================================================ -echo "# part B: the real roots, the real registry (read-only)" -# ============================================================================ -export XDG_RUNTIME_DIR="${HARNESS_TMP_XDG:-/run/user/$(id -u)}" -REAL_REG="$XDG_RUNTIME_DIR/9p" -REAL_SOCKET="$REAL_REG/harness" -if [ -e "$REAL_SOCKET" ]; then - echo "SKIP: a live daemon already owns the posted name `harness`" -else - HOME_DIR=$(getent passwd "$(id -u)" | cut -d: -f6) - start_daemon --name harness - if wait_posted "$REAL_SOCKET"; then - expect_contains "posted name listed in the real registry" harness "$(ls "$REAL_REG")" - expect_contains "ls / shows the claude root" claude "$(p9 "$REAL_SOCKET" ls /)" - # A real transcript, byte-identical through the tree. - REAL_T=$(ls "$HOME_DIR/.claude/projects"/*/*.jsonl 2>/dev/null | head -n 1 || true) - if [ -n "$REAL_T" ]; then - REL="${REAL_T#"$HOME_DIR/.claude/projects/"}" - p9 "$REAL_SOCKET" read "/claude/projects/$REL" > "$TMP/real-via-9p" 2>/dev/null \ - && cmp -s "$TMP/real-via-9p" "$REAL_T" \ - && pass "real transcript byte-identical through the tree" \ - || fail "real transcript byte-identical through the tree" "$REAL_T" - else - echo "skip - no real claude transcript found" - fi - # The credentials at ~/.claude are unreachable: no mount serves the - # root dir, and the exclusion rule holds everywhere else. - p9 "$REAL_SOCKET" read /claude/.credentials.json >/dev/null 2>&1 \ - && fail "live .credentials.json unreachable" "read succeeded" \ - || pass "live .credentials.json unreachable" - CLAUDE_LS=$(p9 "$REAL_SOCKET" ls /claude) - expect_missing "no credentials leaked into /claude" credentials "$CLAUDE_LS" - # The money shot: the mntgen mount every interactive fish sees. - if [ -n "$NS" ]; then - OUT=$(timeout 60 "$NS" --mntgen -- sh -c 'ls /mnt/9p/harness && head -c 200 /mnt/9p/harness/claude/history' 2>"$TMP/mntgen.err") - RC=$? - if [ $RC -eq 0 ]; then - expect_contains "mntgen mount lists the harness tree" claude "$OUT" - expect_contains "mntgen mount reads claude/history" claude "$OUT" - else - fail "mntgen money shot (exit $RC)" "$(cat "$TMP/mntgen.err")" - fi - else - echo "skip - mntgen money shot: 9ns not provided" - fi - kill "$DAEMON_PID" 2>/dev/null - wait "$DAEMON_PID" 2>/dev/null - sleep 0.2 - [ -S "$REAL_SOCKET" ] && fail "stop unposts the real name" "socket still there" || pass "stop unposts the real name" - else - fail "daemon posts into the real registry" "$(cat "$TMP/daemon.log")" - fi -fi - -echo "# $PASSED passed, $FAILED failed" -[ "$FAILED" -eq 0 ] diff --git a/build.zig b/build.zig index 795d8e2..c5d662f 100644 --- a/build.zig +++ b/build.zig @@ -156,7 +156,7 @@ pub fn build(b: *std.Build) void { const is_linux = target.result.os.tag == .linux; const want_9proc = b.option(bool, "9proc", "Build the 9proc library module and demo (default: target is Linux)") orelse is_linux; const want_9ns = b.option(bool, "9ns", "Build the 9ns FUSE mount CLI (default: target is Linux; Linux only)") orelse is_linux; - const want_9harness = b.option(bool, "9harness", "Build the harness fs daemon (default: target is Linux; Linux only)") orelse is_linux; + const want_9agents = b.option(bool, "9agents", "Build the coding-agents fs daemon (default: target is Linux; Linux only)") orelse is_linux; if (want_9ns and !is_linux) { std.debug.print("error: -D9ns=true needs a Linux target (got {s})\n", .{@tagName(target.result.os.tag)}); std.process.exit(1); @@ -207,21 +207,21 @@ pub fn build(b: *std.Build) void { programs_itest.dependOn(p.itest_step); } - // The harness fs daemon: a read-only 9P view of every harness's - // state, posted by default as `harness` (Linux only: it posts into + // The coding-agents fs daemon: a read-only 9P view of every harness's + // state, posted by default as `agents` (Linux only: it posts into // $XDG_RUNTIME_DIR/9p through cloud9.post). - if (want_9harness and !is_linux) { - std.debug.print("error: -D9harness=true needs a Linux target (got {s})\n", .{@tagName(target.result.os.tag)}); + if (want_9agents and !is_linux) { + std.debug.print("error: -D9agents=true needs a Linux target (got {s})\n", .{@tagName(target.result.os.tag)}); std.process.exit(1); } - const harness_build = @import("9harness/build.zig"); - const harness: ?harness_build.Artifacts = if (want_9harness) harness_build.add(b, .{ + const agents_build = @import("9agents/build.zig"); + const agents: ?agents_build.Artifacts = if (want_9agents) agents_build.add(b, .{ .target = target, .optimize = optimize, .cloud9 = module, .ns = if (ns) |p| p.exe else null, }) else null; - if (harness) |p| { + if (agents) |p| { programs_test.dependOn(p.test_step); programs_itest.dependOn(p.itest_step); } diff --git a/build.zig.zon b/build.zig.zon index d042fe3..b275168 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -7,11 +7,11 @@ // published package cannot build: build.zig @imports each program's // build fragment unconditionally, so a program missing from this // list is a FileNotFound for every consumer, while building fine - // from a checkout. 9harness was exactly that until a pardes build + // from a checkout. 9agents was exactly that until a pardes build // caught it. .paths = .{ "build.zig", "build.zig.zon", "src", "web", "test", "docs", "README.md", - "9ns", "9proc", "9harness", + "9ns", "9proc", "9agents", }, } -- cgit v1.3