From df20863879fe2d83077534f4726a985ffc239def Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 22 Sep 2026 10:30:02 -0300 Subject: Rename 9harness to 9agents; README, file-backed qids, worded errors - 9harness/ becomes 9agents/ (build option -D9agents, package paths). - 9agents serves /README, reports qid paths from the file's (dev, ino) and qid versions that move with the file, and names its refusals. --- 9agents/build.zig | 54 ++ 9agents/docs/DESIGN.md | 515 +++++++++++ 9agents/src/active.zig | 1006 ++++++++++++++++++++ 9agents/src/main.zig | 382 ++++++++ 9agents/src/tree.zig | 2413 ++++++++++++++++++++++++++++++++++++++++++++++++ 9agents/test/e2e.sh | 358 +++++++ 6 files changed, 4728 insertions(+) create mode 100644 9agents/build.zig create mode 100644 9agents/docs/DESIGN.md create mode 100644 9agents/src/active.zig create mode 100644 9agents/src/main.zig create mode 100644 9agents/src/tree.zig create mode 100755 9agents/test/e2e.sh (limited to '9agents') diff --git a/9agents/build.zig b/9agents/build.zig new file mode 100644 index 0000000..0a63d9b --- /dev/null +++ b/9agents/build.zig @@ -0,0 +1,54 @@ +//! Build fragment for 9agents: the coding-agents fs daemon (binary `9agents`), +//! a read-only fresh-from-disk 9P2000 view of every harness's state, +//! posted by default under the name `agents`. It is `@import`ed by the +//! root build.zig and called with the root builder, so every `b.path(...)` +//! here is relative to the cloud9 root (hence the `9agents/` prefix), +//! every option is defined by the root and every step it registers lands +//! in the root's step list under the `9agents` prefix. +//! +//! Steps: 9agents, 9agents-test, 9agents-itest. +const std = @import("std"); + +pub const Context = struct { + target: std.Build.ResolvedTarget, + optimize: std.builtin.OptimizeMode, + cloud9: *std.Build.Module, + /// The 9ns binary, for the --mntgen money shot in the integration + /// suite; null when 9ns is disabled, in which case the mntgen + /// section of the suite is skipped. + ns: ?*std.Build.Step.Compile, +}; + +pub const Artifacts = struct { + exe: *std.Build.Step.Compile, + /// `9agents-test`: the tree's unit tests (exclusions, path safety, + /// node ids, vanishing files — all against a fake HOME). + test_step: *std.Build.Step, + /// `9agents-itest`: test/e2e.sh (fixture roots + scratch registry, + /// plus the real-registry end-to-end when the `agents` name is free). + itest_step: *std.Build.Step, +}; + +pub fn add(b: *std.Build, ctx: Context) Artifacts { + const mod = b.createModule(.{ + .root_source_file = b.path("9agents/src/main.zig"), + .target = ctx.target, + .optimize = ctx.optimize, + .imports = &.{.{ .name = "cloud9", .module = ctx.cloud9 }}, + }); + const exe = b.addExecutable(.{ .name = "9agents", .root_module = mod }); + const install = b.addInstallArtifact(exe, .{}); + b.getInstallStep().dependOn(&install.step); + b.step("9agents", "Build and install only the coding-agents fs daemon").dependOn(&install.step); + + const test_step = b.step("9agents-test", "Run the 9agents tree's unit tests (fake HOME; never the live roots)"); + test_step.dependOn(&b.addRunArtifact(b.addTest(.{ .root_module = mod })).step); + + const itest_step = b.step("9agents-itest", "Run 9agents/test/e2e.sh (posted name, reads, cmp of a transcript, mntgen mount, exclusions, live appends)"); + const run = b.addSystemCommand(&.{"bash"}); + run.addFileArg(b.path("9agents/test/e2e.sh")); + run.addArtifactArg(exe); + if (ctx.ns) |ns| run.addArtifactArg(ns); + itest_step.dependOn(&run.step); + return .{ .exe = exe, .test_step = test_step, .itest_step = itest_step }; +} diff --git a/9agents/docs/DESIGN.md b/9agents/docs/DESIGN.md new file mode 100644 index 0000000..8d399d3 --- /dev/null +++ b/9agents/docs/DESIGN.md @@ -0,0 +1,515 @@ +# 9agents design + +9agents is the coding-agents fs daemon: a long-running 9P2000 server that +replaces the `zmxify` rc script's introspection half with a unified, +file-shaped, always-fresh view of every AI-agent harness's state on the +machine — Claude Code (`~/.claude`), Codex (`~/.codex`), omp (`~/.omp`), +hermes (`~/.hermes`) and dsh (`~/.dsh`), plus their skills. Everything is +read-only and read live from disk at request time; nothing is cached, so +a session transcript grows as its harness writes it and a new session +appears as soon as its file lands. + +One binary, hosted Linux only (it posts through `cloud9.post`). The +backend is `src/tree.zig`, a `cloud9.fs.Server` backend run by +`cloud9.serve.Runner`; the CLI is `src/main.zig`. Like its siblings +(9proc, 9ns) the build is a fragment in `build.zig` wired into the root +behind `-D9agents` (steps `9agents`, `9agents-test`, `9agents-itest`; +installed by the plain `zig build` next to the other programs). + +## The tree (v1 contract) + +``` +/ read-only root (dr-xr-xr-x) +/README the tree explained in place: what each entry is, + the /active fields, and the exclusions. Static text, + so it is the one fact needing no buffer. +/pid the daemon's pid, one line +/uptime seconds since start, one line +/claude/ + projects/ mirror of ~/.claude/projects//: one dir + per project, its *.jsonl session transcripts as plain + readable files, memory/ subdirs included + history ~/.claude/history.jsonl (global prompt history) + skills/ mirror of ~/.claude/skills +/codex/ + sessions/ mirror of ~/.codex/sessions/YYYY/MM/DD/rollout-*.jsonl + session-index ~/.codex/session_index.jsonl + history ~/.codex/history.jsonl +/omp/ mirror of ~/.omp/agent (history.db, agent.db, models.db, + ... served as raw readable blobs; sqlite parsing is a + later step — v1 is files) +/hermes/ mirror of ~/.hermes (state.db raw, logs/) +/dsh/ mirror of ~/.dsh (profiles/, storages/, ... whatever it holds) +/skills/ union view: plain dirs claude/, codex/, omp/ mirroring each + harness's skills dir; a harness without one simply has no entry +``` + +Mirroring is lazy: nothing is walked at startup. A lookup, getattr or +readdir stats and lists the real files at request time. Unknown files and +directories appear as themselves, readable. A file that vanishes between +lookup and read answers ENOENT cleanly; a file that appears is visible on +the next request. + +Writes, creates, removes and setattrs answer EPERM (create/remove/wstat +are not declared as backend features, so the engine refuses them itself; +the write and setattr ops answer `E.PERM`, as does an open for write). +Modes are reported read-only regardless of the real bits: directories +`0o555`, files `0o444`. Sizes and mtimes are real. + +The engine's known fidelity gap applies: directory records carry fixed +modes and length 0, so `9p ls -l` shows placeholders; `9p stat` is the +accurate one. + +## The exclusions (the security boundary) + +The daemon must never become a credential reader for anything that +mounts it. The rule is absolute — `excluded()` in tree.zig, unit-tested — +and applies to every path component of every subtree, by lookup *and* by +readdir, so an excluded name neither resolves nor lists: + +- names containing `credentials`, `token`, `auth` or `secret` + (case-insensitive substrings; "auth" also hides "author-notes" — the + price of never guessing wrong); +- names ending `.key`, `.pem` or `.env`; +- the config/settings files of the mirrored harnesses, which may embed + API keys: `settings.json`, `settings.yaml`, `settings.local.json`, + `config.yml`, `config.yaml`, `config.toml`, `models.yml`, `models.yaml` + (Claude Code's settings.json is not in the tree anyway: /claude serves + only projects/, skills/ and history.jsonl); +- `.ssh`, and `id_rsa`/`id_dsa`/`id_ecdsa`/`id_ed25519` prefixes. + +Symlinks are never served, at any depth: they are an escape hatch around +the root pinning (a symlink out of `~/.claude` would make the daemon read +outside the five roots). A symlink answers ENOENT and is not listed. + +Path resolution is the mechanism that makes that true, and it is not +string composition. `openIn` walks from the pinned base one component at +a time, each opened with `O_NOFOLLOW`: the intermediate components with +`O_PATH|O_DIRECTORY` (so a component swapped for a symlink fails with +ENOTDIR rather than redirecting the walk) and the last with the caller's +flags. Every stat, read and readdir goes through it. Composing +`/` and opening that in one call is what the first version +did, and it was wrong in a way a test could not see: only the final +component's symlink was refused, so a name replaced between the walk and +the read — the plain TOCTOU — served bytes from outside every root, and +a directory above it could redirect the whole path at any time. `O_PATH` +also means a fifo or device left in a harness root is stat'd without its +open ever blocking; a read refuses anything that is not a regular file. + +The base itself comes from $HOME or `--root NAME=PATH` at startup and is +resolved normally (it is configuration, and may legitimately be a link). +Relative paths are composed only from remembered (root, rel) pairs plus +single-segment, engine-vetted names. Client bytes never form a path: +names with `/` or NUL are refused before the filesystem is touched, `.` +and `..` are refused as components inside the walk, and `..` as a lookup +resolves through the backend's own parent map, never the kernel's. + +## Qids, and what the protocol says they mean + +Read out of the Plan 9 source (`~/05-genizah/principia-softwarica`), not +recalled — the three rules below were all being broken. + +`intro(5)`: "The qid represents the server's unique identification for the +file being accessed: two files on the same server hierarchy are the same if +and only if their qids are the same. ... The path is an integer unique among +all files in the hierarchy. If a file is deleted and recreated with the same +name in the same directory, the old and new path components of the qids +should be different. The version is a version number for a file; typically, +it is incremented every time the file is modified." + +* **The qid path is the file's identity, not the server's bookkeeping.** The + backend's node id is a slot in the path table, and a slot freed by its last + release returns with a new generation — the guard that stops a forged node + id resolving. Reporting that as the qid path meant `9p stat` of one + unchanged file answered a *different* path every time (`...50000104`, + `...70000104`, `...90000104`), which by the rule above says "a different + file" on every walk. The qid path now comes from the file itself, the way + u9fs builds one — `qid.vers = st->st_mtime ^ (st->st_size << 8)` and a path + from the inode with the device mixed in (`u9fs.c`) — while the engine goes + on resolving by node id. `fs.Attr.path` exists for exactly this split and + the engine uses it for nothing else. The device is folded in because an + inode number is only unique within its filesystem and the five roots need + not share one. + +* **The qid version has to move when the file does.** It was 0 everywhere, + which tells every client that nothing this server holds ever changes — + precisely wrong for a tree whose whole point is transcripts that grow while + you read them. It is now `mtime ^ (size << 8)` for a file on disk (u9fs's + formula: the shift is what makes an append within one second still change + it) and a hash of the text for the fields this server derives rather than + reads, where `status` flipping `busy` to `idle` keeps its size and only the + content can carry the change. + +* **A directory's length is 0 on purpose.** `stat(5)`: "Directories and most + files representing devices have a conventional length of 0." That part of + the engine's "fidelity gap" note below is the convention, not a shortfall. + +## Errors are strings + +`error(5)` is `Rerror tag[2] ename[s]`: 9P2000 has no numeric error, and acme +names its refusals in words (`Eperm[] = "permission denied"`, `Eexist[] = +"file does not exist"`, `editors/acme/fsys.c`). A server that answers only +bare errnos throws away the one channel the protocol gives it for explaining +itself, and a move has seven different reasons to refuse that all used to +arrive as one undifferentiated `EPERM`: moves not allowed, illegal name, no +session resolved, no cwd, dsh has no resume form, nothing to resume, the name +is taken. Each says so now, and so does the over-cap listing, which used to +claim "Too many open files in system" — a true sentence about the wrong +resource. + +The errno is still carried beside the text, because the reader may be a Linux +mount rather than a person: 9ns maps the *text* back to an errno by +case-insensitive substring, first match wins (`enameToErrno`, 9ns/src/nine.zig). +So each message keeps the phrase that carries its errno — "not permitted" for +EPERM, "invalid" for EINVAL, "exists" for EEXIST, "no such" for ENOENT — and +avoids the ones that would silently retarget it: "permission" and "denied" map +to EACCES, "read-only" to EROFS, and the bare substring "fid" to EBADF. That +is why the read-only refusal reads "9agents serves a view, not a store: +writing is not permitted" and not the more obvious wording. + +Opens refused by the mode bits never reach the backend at all: the engine +checks `perm` first and answers `e_perm`, which is already acme's spelling. + +## Backend shape + +`Harness` (tree.zig) is the backend of `fs.Server(Harness, opts)` on +`serve.Runner` (main.zig): allocation-free request paths, comptime caps, +`Io` passed explicitly, static memory in .bss (the path table and the +per-connection buffers; ~3 MB of tables, untouched pages cost nothing). + +- **Node ids** pack `Node{idx: u8, kind: u8, serial: u48}` in the u64, + zmx-style. Static skeleton nodes (facts, the harness dirs, skills) are + `kind = .top`; every mirrored file is `kind = .path` with + `serial = (slot, generation)` of its table entry. Two walks of the same + file find the same entry, so their node ids — and qid paths — are + equal; an entry freed by its last release takes a new generation, so a + stale forged node id never resolves. +- **The path table** remembers at most 4096 remembered (root, rel) pairs, + each with a Wyhash of the pair for dedupe and a reference count. The + backend declares `features = .{ .references = true }`, so the engine + asks lookups for `.` too and pays every reference back with exactly one + `release`; an entry's refcount reaching zero frees it. Readdir records + name entries without references (a listing does not pin), so a later + walk of the same name finds the same entry and the same node id. A full + table answers `E.NFILE` — on a lookup, and on a listing that cannot + name all its entries (see Readdir). +- **Fresh reads**: a read opens the file, `readPositionalAll`s the window + at the request's offset, and closes it — every read is against the live + file, and the engine handles offsets, so `cat` of a 5 MB transcript + through a mntgen mount works. +- **Readdir** collects a directory's surviving names, sorts them for + cross-read cursor stability, and stages the engine's records + (`node:u64le dir:u8 len:u8 name`), skipping `req.off` records. The + comptime caps (1024 names, 128 KiB of name bytes, the path table) are + **loud**: a directory that would not fit answers `E.NFILE` instead of a + short listing, because a short listing cannot be told apart from a + small directory and is therefore a wrong answer, not a limitation. The + staging buffer filling is different and normal — it ends that read at + the last record that fit, and the next read continues from there; + staging stops at the first record that does not fit rather than packing + a later, shorter one behind it, which would drop that entry from the + listing entirely. A directory that changes between the reads of one + listing may shift its cursor — the fresh-tree trade, accepted in v1. +- **Locking**: one mutex spans `handle` and the reply (the answer's bytes + point into shared staging buffers), so the backend is serialized across + connections. v1 accepts this: it is an introspection fs, not a + throughput service. + +## CLI and process model + +``` +usage: 9agents [--unix PATH | --tcp IP:PORT | --fd N] [--no-post] + [--name NAME] [--root NAME=PATH]... +``` + +- Default: post itself under the name `agents` via + `serve.Runner.listenPosted`, so it appears at + `$XDG_RUNTIME_DIR/9p/agents` and every interactive fish (self-wrapped + in `9ns --mntgen`) sees it at `/mnt/9p/agents` with zero + configuration. `stop()` unposts; the stale-socket protocol covers a + killed daemon. +- `--unix`/`--tcp` listen forms (beside the post, or with `--no-post`), + `--name` to post under another name, `--root NAME=PATH` to pin any of + the five roots elsewhere (defaults are `$HOME/.`; the tests use + fake homes and never the live roots). +- `--fd N` serves exactly one 9P session over the connected stream on + descriptor N (socket activation, `9ns --spawn` handoffs), driving the + engine directly; it posts nothing. +- SIGTERM/SIGINT stop cleanly (unpost); SIGPIPE is ignored. +- No daemonization in v1: the daemon never forks or detaches, so it is + run under something that supervises it. On this machine that is a + systemd user service (`~/.config/systemd/user/9agents.service`, + `Restart=always`, lingering on), which is what keeps `/mnt/9p/agents` + there across logouts and reboots; a restart reclaims the posted name + through the stale-socket protocol. For a throwaway instance the zmx + way still works — `zmx run agents -d 9agents`. + +## Integration test plan (`test/e2e.sh`) + +Part A runs entirely on fixtures (a fake home pinned by `--root`, a +scratch `XDG_RUNTIME_DIR` registry, plan9port's `9p` as the client): + +1. the posted name is listed in the registry; +2. `9p ls /` shows the roots; a fixture transcript reads back + byte-identical (`cmp` against the source file); the skills union + mirrors the harness tree; +3. credentials-shaped files are unreachable by both listing and reading, + at several depths, in every root; +4. a file appended after the daemon started is visible immediately, and + a new session directory appears in the next listing; +5. the pid fact answers the daemon's pid; writes answer EPERM; +6. the `--unix` and `--fd` listen forms serve; +7. SIGTERM unposts the name. + +Part B is the money shot, read-only against the real roots and the real +`$XDG_RUNTIME_DIR/9p` — only when the `agents` name is free (a live +daemon owning the name skips it, not fails): the posted name in the real +registry, `9p` walks of the live `~/.claude`, a real transcript +byte-identical through the tree, the live `~/.claude/.credentials.json` +unreachable, the `9ns --mntgen` mount of `/mnt/9p/agents` with +`head` of `/claude/history`, and the stop unposting the real name. + +Unit tests (tree.zig) cover the exclusions (including the predicate +itself, spelled out), path-composition safety (slashes, NULs, symlinks, +the `..` chain, NOTDIR), node id packing (same file equal, distinct +files differ, the union identity, id recycling through release), the +vanishing file (ENOENT on lookup, getattr and read), EPERM on +write/setattr/open-for-write, and live visibility — all against a fake +HOME under a test tmp dir, never the real harness roots. + +## Verification + +- `zig build` (installs `9agents` beside 9ns, 9proc-demo, 9web). +- `zig build test` — the library's 80 tests stay green. +- `zig build 9agents-test` — the tree's unit tests. +- `zig build 9agents-itest` — `test/e2e.sh` (fixture + real registry). +- `zig build 9proc-check-freestanding` — the daemon is hosted-only but + must not break the root module. +- `zig build programs-test` / `programs-itest` — the umbrella steps, + which include 9agents's. + +## /active: the derived view (v2) + +v1 mirrors files. `/active` is the first *semantic* layer: one directory +per live agent, normalized across harnesses, so `ls /active` answers +"what is running right now" whatever wrote it. It is the discovery half +of `~/.local/bin/zmxify`, whose resolution ladder it ports. + +``` +/active/ + claude/ + 345104/ + pid 345104 ppid 344980 + started 1790013029 cwd /home/goblin + name goblin-e5 title the session's title + session 1f9a74f7-... model claude-opus-5[1m] + via registry zmx harness (read-write, below) + status busy only where the harness publishes one + transcript the live .jsonl, growing as it writes + agents/ /{model,transcript} + omp/ + 155574/ ... +``` + +A harness directory holds one directory per live process of it, named +by pid. `/proc` spells it that way and so does this: a compound +`claude-345104` would make you parse a name to recover `harness`, which +is already the directory above it, and `/active/omp/*` would not glob. + +There is no `updated` file. The last write to a session is the mtime of +`transcript`, which `stat` already carries; serving it again as its own +file would be the same fact twice, and the copy is the one that goes +stale. + +These are zmxify's picker columns — pid, harness, dir, age, zmx, via, +session, title — as files, which is the point: the script stops +scanning `/proc`, reading fds and querying sqlite itself and just reads +the tree. + +**`status` is a promise only some harnesses make.** It is the harness's +own word for what it is doing, and only Claude Code publishes one +(`busy`, in `sessions/.json`); for every other harness the file is +simply absent, the way a `skills` mount with no target is absent. +Deriving one from the process's `/proc` state would answer a different +question (`S` means "not on a CPU this instant", not "waiting for +you"). For every other harness the honest activity signal is the mtime +of `transcript`. `ls /active/*/*/status` tells you who publishes the +stronger answer. + +An entry is named `-`: unique, stable for the process's +life, and it sorts by harness. The pid is the identity because it is +what `/proc` and every harness's own registry agree on. + +### Finding the agents + +`/proc` is scanned for a process whose `argv[0]` basename is `omp`, +`claude`, `codex`, `hermes` or `dsh`, or a python running +`hermes_cli.main`. A command line carrying one of the daemon words +(`gateway`, `dashboard`, `mcp`, `mcp-server`, `app-server`, +`exec-server`, `serve`, `daemon`, `acp`, `ps`, `render`, `export`, +`__omp_worker_daemon_broker`) is a server or a helper, never a session. +The daemon's own ancestry is excluded, so 9agents can never list or +act on the process tree it lives in. + +Liveness is `/proc/` existing **and** its `stat` field 22 +(starttime) matching the one remembered for the slot. A pid that has +been reused is a different process and does not list; a corpse never +lists. + +### Resolving the session + +Each harness is asked in its own terms — the `fd -> dir -> db` ladder +zmxify worked out, with the route named in `via` so a wrong guess is +visible rather than silent: + +| harness | route | `via` | +|---|---|---| +| claude | `~/.claude/sessions/.json`, which the harness maintains itself: sessionId, cwd, name, status, version, and `procStart` as a pid-reuse guard | `registry` | +| omp | the transcript it holds open under `~/.omp/agent/sessions/`, newest first; else the store directory named after its cwd with `/` becoming `-` | `fd`, `dir` | +| dsh | the `session-/session.jsonl.zstd` it holds open; zstd, so there is no title | `fd` | +| codex | the rollout it holds open — `sessions/YYYY/MM/DD/rollout--.jsonl`, whose *name* carries the session id, so its sqlite is never opened | `fd` | +| hermes | nothing to resolve: see below | `none` | + +A resolved session is checked against the process's cwd before it is +believed (the head of an omp transcript carries `"cwd"`). A process +whose session does not resolve still appears, with everything `/proc` +knows and an empty `session`: the view never pretends to know what it +does not, and a move on it is refused. + +codex's id is the last 36 bytes of the rollout's name, not what +splitting on `-` gives — the timestamp in front of it holds dashes too. + +**hermes is the one harness with no answer here, and it needs none.** +Its sessions live only in `state.db`, with no per-session file to find; +but the only hermes processes that run are the gateway and the +dashboard, and both are daemon-shaped, so neither is a session anything +should move. zmxify resolved a hermes id from sqlite and then declined +to touch the process holding it, for the same reason. If an interactive +hermes ever exists, this is the gap, and it is the one place sqlite +would buy something. + +### Freshness + +A slot remembers only what identifies the agent: harness, pid, +starttime, cwd, session id and transcript path. Every *field* is +re-derived from disk when it is read, so `status` is never stale and a +transcript grows under `cat`. `/proc` is rescanned on a readdir of +`/active` and on a lookup that misses, not on every read. + +### `zmx`: the write path, and why it is not a ctl + +Writing a zmx session name into `/active//zmx` moves that agent into +a zmx session of that name: the zmxify action half, as a file. + +Reading `zmx` gives the `ZMX_SESSION` of the process, empty when it runs +outside zmx. Writing sets it. The file means *which zmx session this +agent lives in*, and writing a name makes that true — state, not a verb +channel. A `ctl` taking words would be the ordinary Plan 9 spelling +(`/proc/n/ctl`), and an executable `zmxify` script served in the tree +would be the zmx `attach` spelling, but a script that shells out to a +local binary is a lie over a remote mount: it would run against a +session that is not on the client's machine. A write is served where +the authority is, so it survives being mounted from anywhere. + +What a write does, in order, refusing before it destroys anything: + +1. The name must be zmx's label charset (`[A-Za-z0-9._-]`) and unused by + a live session, else `EEXIST`. +2. The agent's session must have resolved, else `EPERM`. Nothing is + killed that has nowhere to come back to — zmxify's invariant. +3. `SIGTERM`, then `SIGKILL` after 5s, giving up at 12s with `EIO`. +4. `fork`, `chdir` to the agent's cwd, `exec zmx run -d` with a + **fixed argv per harness** (`claude --resume `, + `omp --resume `, `codex resume `, + `hermes --resume `). No client byte ever reaches `exec`: the + only thing the client supplies is the session name, and it is + validated first. +5. The write returns once `$XDG_RUNTIME_DIR/9p/zmx/` appears + (zmx self-posts), or `EIO` on timeout. + +The agent comes back under a new pid, so the old `/active/` is gone +and a new entry takes its place with `zmx` reading the new name. The +window between the kill and the exec is the same exposure zmxify has +always had, and it is why step 2 comes first. + +**This is the tree's first write path, and it is an escalation**: a +client that can write this file can kill the user's agents and cause a +process to be spawned. It is therefore off unless `--allow-move` is +given, and the read-only daemon stays the default. Everything else in +the tree still answers `EPERM` to writes. + +### Where this is not Plan 9 + +Named, because they are choices rather than oversights: + +* **`via` describes how the server found out**, not something true of + the process. That is diagnostics in the interface. It stays because a + resolution that guesses wrong silently is worse than a wart — it is + why zmxify's picker showed the route too. +* **A write to `zmx` does not change the object, it replaces it.** The + process is killed and another starts under a new pid, so the entry + written to disappears. `/proc/n/ctl` accepting `kill` is at least + honest about being a verb with a consequence. The file still beats an + executable served in the tree, which would lie outright over a remote + mount, so it stays — with this paragraph. +* **A move blocks the whole daemon** for as long as the kill and the + restart take, because one mutex spans `handle` and the reply. A Plan + 9 server keeps answering other fids meanwhile. The engine already has + parked replies and Tflush for exactly this; the move does not use + them yet, and that is the first thing to revisit. +* **`/active` lives inside the mirror** rather than being its own + posted service. "What files exist" and "what is running" are + different jobs, and a stricter reading would separate them; they + share the pinned roots and the parsing glue, so they share a daemon. + +### Replacing zmxify + +There is no replacement script, because the mount is the replacement: + +``` +ls /mnt/9p/agents/active/*/* what is running +cat /mnt/9p/agents/active/claude/345104/session what it is +echo mine > /mnt/9p/agents/active/claude/345104/zmx +zmx attach mine +``` + +`~/.local/bin/zmxify` was 307 lines of rc because there was no +filesystem to ask: it scanned `/proc`, classified command lines, read +fd tables and queried two sqlite databases to work out what the four +lines above now read. All of that moved into the daemon, where it is +tested. Writing a wrapper back on top would only re-derive what the +tree already says, and would become a second, worse interface beside +the real one. + +Two behaviours the old script had are worth knowing, because they are +now the caller's to arrange and both are one line: pick a session name +nothing holds (the registry, `$XDG_RUNTIME_DIR/9p/zmx`, says which are +taken) and `zmx attach` afterwards. + +A third is gone rather than moved. The old script refused to act on +its own ancestry, because it did the killing itself: kill your own +parent and the script dies before it can re-exec, losing the session it +was rescuing. The daemon kills, re-execs and waits for the new session +to post, and completes all of it whether or not the client is still +connected — verified by hanging up immediately after sending the write +and watching the move land anyway. So moving the terminal you are +sitting in works, which is the common case; the shell drops, and you +reattach with the name you wrote. + +### Testability + +`--proc DIR` overrides `/proc` the way `--root NAME=PATH` overrides a +harness root, so the scan, the liveness guard, the resolution ladder and +the exclusions are unit-tested against a fixture process tree and never +against the live machine. The move itself is exercised end-to-end +against a fake harness in `test/e2e.sh`. + +## Out of scope for v1 (documented, not hidden) + +- Write paths (create/write/setattr stay EPERM), resume/attach + automation (the zmxify re-exec half), sqlite parsing (omp/hermes + sessions stay raw blobs), cross-session search, any caching or + invalidation layer. +- Streaming/parked reads (a read answers at once; the transcripts are + plain files), and union-directory semantics beyond the plain + /skills/ dirs. diff --git a/9agents/src/active.zig b/9agents/src/active.zig new file mode 100644 index 0000000..daaddc9 --- /dev/null +++ b/9agents/src/active.zig @@ -0,0 +1,1006 @@ +//! The derived view behind `/active`: one record per live agent, +//! normalized across harnesses. +//! +//! v1 of the tree mirrors files. This module answers a different +//! question — *what is running right now* — by reading `/proc` and then +//! asking each harness in its own terms which stored session the +//! process is writing. That ladder (`fd` -> `dir` -> `db`, with the +//! route named so a wrong guess is visible) is `~/.local/bin/zmxify`'s, +//! ported here so the knowledge lives somewhere tested. +//! +//! Everything here composes paths from the proc root, a pinned harness +//! root and a pid. No client byte reaches the filesystem, and the proc +//! root is a parameter (`--proc`) so the scan, the liveness guard and +//! the exclusions are testable against a fixture tree. +const std = @import("std"); +const Io = std.Io; + +// ---- comptime bounds --------------------------------------------------------- + +/// Live agents held at once. A machine running more than this many +/// interactive harnesses at the same time is not the case to optimize. +pub const max_live: usize = 32; +/// Subagents listed under one agent. +pub const max_agents: usize = 32; +/// Longest directory a process may run in. +pub const cwd_capacity: usize = 512; +/// Longest absolute path this module composes (proc root, harness root). +pub const path_capacity: usize = 1024; +/// Longest session id, title, name or model answered. +pub const text_capacity: usize = 160; +/// `-`. +pub const name_capacity: usize = 32; +/// Bytes of a `/proc` file or a transcript head read at once. +pub const probe_capacity: usize = 8192; + +/// The harnesses this view knows how to recognize. +/// The order is `tree.Root`'s, and `tree.zig` asserts that at comptime: +/// the two enums index the same pinned roots, so a mismatch would hand +/// one harness another's base path. +pub const Kind = enum(u8) { + claude, + codex, + omp, + hermes, + dsh, + + pub fn text(k: Kind) []const u8 { + return @tagName(k); + } +}; + +/// How a session was resolved — reported, so a wrong guess is visible +/// rather than silent. `none` means the process is listed with what +/// `/proc` knows and nothing more, and cannot be moved: hermes is the +/// only harness that always lands here. +pub const Via = enum(u8) { + none, + registry, + fd, + dir, + + pub fn text(v: Via) []const u8 { + return @tagName(v); + } +}; + +/// A fixed-capacity byte buffer. Everything this module remembers is +/// bounded at comptime, like the rest of the daemon. +fn Buf(comptime n: usize) type { + return struct { + bytes: [n]u8 = undefined, + len: u16 = 0, + + const Self = @This(); + + pub fn set(b: *Self, v: []const u8) void { + const take = @min(v.len, n); + @memcpy(b.bytes[0..take], v[0..take]); + b.len = @intCast(take); + } + + pub fn slice(b: *const Self) []const u8 { + return b.bytes[0..b.len]; + } + + pub fn clear(b: *Self) void { + b.len = 0; + } + }; +} + +/// One live agent. +pub const Live = struct { + kind: Kind = .claude, + pid: u32 = 0, + ppid: u32 = 0, + /// `/proc//stat` field 22. Two processes can share a pid over + /// time; they cannot share a pid and a start time, so this is what + /// makes a remembered slot safe to trust later. + starttime: u64 = 0, + /// Unix seconds, from the process's start time against boot time. + started: i64 = 0, + via: Via = .none, + name: Buf(name_capacity) = .{}, + cwd: Buf(cwd_capacity) = .{}, + session: Buf(text_capacity) = .{}, + /// Absolute path of the session transcript, empty when unresolved. + transcript: Buf(path_capacity) = .{}, + /// Directory holding the subagent transcripts, empty when the + /// harness has no such layout. + agent_dir: Buf(path_capacity) = .{}, +}; + +// ---- pure helpers: the part that carries zmxify's knowledge ------------------- + +/// A command line word that marks a daemon, a server or a helper — a +/// process wearing the harness's name that is not an interactive +/// session and must never be listed or acted on. +const not_session = [_][]const u8{ + "__omp_worker_daemon_broker", + "acp", + "mcp", + "mcp-server", + "app-server", + "exec-server", + "gateway", + "dashboard", + "serve", + "daemon", + "ps", + "render", + "export", +}; + +/// The last path component of `p`. +pub fn basename(p: []const u8) []const u8 { + if (std.mem.lastIndexOfScalar(u8, p, '/')) |i| return p[i + 1 ..]; + return p; +} + +/// Iterates a `/proc//cmdline`: NUL-separated argv, usually with a +/// trailing NUL. An empty cmdline (a kernel thread) yields nothing. +pub const Argv = struct { + rest: []const u8, + + pub fn init(cmdline: []const u8) Argv { + return .{ .rest = cmdline }; + } + + pub fn next(a: *Argv) ?[]const u8 { + while (a.rest.len > 0 and a.rest[0] == 0) a.rest = a.rest[1..]; + if (a.rest.len == 0) return null; + const end = std.mem.indexOfScalar(u8, a.rest, 0) orelse a.rest.len; + const word = a.rest[0..end]; + a.rest = a.rest[@min(end + 1, a.rest.len)..]; + return word; + } +}; + +/// The harness a command line runs, if any. `argv[0]`'s basename names +/// it, except for hermes, which runs as a python module; any word from +/// `not_session` disqualifies the whole command line. +pub fn harnessOf(cmdline: []const u8) ?Kind { + var it: Argv = .init(cmdline); + const argv0 = it.next() orelse return null; + const exe_name = basename(argv0); + + var found: ?Kind = null; + if (std.mem.eql(u8, exe_name, "claude")) { + found = .claude; + } else if (std.mem.eql(u8, exe_name, "omp")) { + found = .omp; + } else if (std.mem.eql(u8, exe_name, "codex")) { + found = .codex; + } else if (std.mem.eql(u8, exe_name, "hermes")) { + found = .hermes; + } else if (std.mem.eql(u8, exe_name, "dsh")) { + found = .dsh; + } else if (std.mem.startsWith(u8, exe_name, "python") or std.mem.startsWith(u8, exe_name, "pypy")) { + var py: Argv = .init(cmdline); + while (py.next()) |w| { + if (std.mem.endsWith(u8, w, "hermes") or std.mem.endsWith(u8, w, "hermes_cli.main")) { + found = .hermes; + break; + } + } + } + if (found == null) return null; + + var words: Argv = .init(cmdline); + while (words.next()) |w| { + for (not_session) |bad| if (std.mem.eql(u8, w, bad)) return null; + } + return found; +} + +/// Field `n` (1-based) of a `/proc//stat` line. Field 2 is the +/// executable name in parentheses and may itself hold spaces and +/// parentheses, so the fields are counted from the *last* `)`, never by +/// splitting the whole line. Only fields from 3 on are reachable, which +/// is all any caller wants. +pub fn statField(stat: []const u8, n: usize) ?u64 { + if (n < 3) return null; + const close = std.mem.lastIndexOfScalar(u8, stat, ')') orelse return null; + var rest = stat[close + 1 ..]; + var field: usize = 3; // state, the first field after the name + while (true) { + while (rest.len > 0 and rest[0] == ' ') rest = rest[1..]; + if (rest.len == 0) return null; + const end = std.mem.indexOfScalar(u8, rest, ' ') orelse rest.len; + if (field == n) return std.fmt.parseInt(u64, rest[0..end], 10) catch null; + rest = rest[end..]; + field += 1; + } +} + +/// The process's start time in clock ticks since boot. Two processes can +/// share a pid over time; they cannot share a pid and a start time, so +/// this is what makes a remembered slot safe to trust later. +pub fn startTimeOf(stat: []const u8) ?u64 { + return statField(stat, 22); +} + +/// The parent pid. +pub fn parentOf(stat: []const u8) ?u32 { + const v = statField(stat, 4) orelse return null; + return std.math.cast(u32, v); +} + +/// The value of `"key"` in a small flat JSON object: a quoted string +/// without its quotes, or a bare token (a number, `true`, `null`). +/// Nested objects are not searched, which is what the callers want — +/// these are the harnesses' own one-level session records. +pub fn jsonField(text: []const u8, key: []const u8) ?[]const u8 { + var quoted_buf: [64]u8 = undefined; + if (key.len + 2 > quoted_buf.len) return null; + quoted_buf[0] = '"'; + @memcpy(quoted_buf[1 .. 1 + key.len], key); + quoted_buf[1 + key.len] = '"'; + const quoted = quoted_buf[0 .. key.len + 2]; + + var from: usize = 0; + while (std.mem.indexOfPos(u8, text, from, quoted)) |at| { + from = at + quoted.len; + var rest = text[from..]; + while (rest.len > 0 and (rest[0] == ' ' or rest[0] == '\t')) rest = rest[1..]; + if (rest.len == 0 or rest[0] != ':') continue; + rest = rest[1..]; + while (rest.len > 0 and (rest[0] == ' ' or rest[0] == '\t')) rest = rest[1..]; + if (rest.len == 0) return null; + if (rest[0] == '"') { + const end = std.mem.indexOfScalar(u8, rest[1..], '"') orelse return null; + return rest[1 .. 1 + end]; + } + if (rest[0] == '{' or rest[0] == '[') continue; // not a leaf; keep looking + const end = std.mem.indexOfAny(u8, rest, ",}\n\r \t") orelse rest.len; + if (end == 0) return null; + return rest[0..end]; + } + return null; +} + +/// The session id in a transcript's file name. omp writes +/// `_.jsonl`, so the id is what follows the last `_`. +/// codex writes `rollout--.jsonl`, where the +/// timestamp holds dashes too, so the id is the last 36 bytes rather +/// than anything found by splitting. Claude Code writes +/// `.jsonl`, all of it the id. +pub fn sessionIdOf(kind: Kind, file_name: []const u8) []const u8 { + var stem = file_name; + if (std.mem.endsWith(u8, stem, ".jsonl")) stem = stem[0 .. stem.len - ".jsonl".len]; + switch (kind) { + .omp => if (std.mem.lastIndexOfScalar(u8, stem, '_')) |i| return stem[i + 1 ..], + .codex => { + const uuid_len = "01a0bcd2-5653-7cc0-aa36-676f6467a72a".len; + if (stem.len >= uuid_len) return stem[stem.len - uuid_len ..]; + }, + else => {}, + } + return stem; +} + +/// The store directory a harness derives from a working directory. +/// omp strips `$HOME` and turns every `/` into `-`; Claude Code turns +/// every character that is not a letter or a digit into `-`, over the +/// whole path. Returns the slug written into `out`. +pub fn slugOf(kind: Kind, cwd: []const u8, home: []const u8, out: []u8) ?[]const u8 { + var path = cwd; + if (kind == .omp and home.len > 0 and std.mem.startsWith(u8, path, home)) { + path = path[home.len..]; + } + if (path.len > out.len) return null; + for (path, 0..) |c, i| { + out[i] = switch (kind) { + .omp => if (c == '/') '-' else c, + else => if (std.ascii.isAlphanumeric(c)) c else '-', + }; + } + return out[0..path.len]; +} + +/// Is `name` a zmx session name? zmx allows only these bytes in a +/// label, and the move path never composes anything else. +pub fn legalZmxName(name: []const u8) bool { + if (name.len == 0 or name.len > 64) return false; + for (name) |c| { + if (!std.ascii.isAlphanumeric(c) and c != '.' and c != '_' and c != '-') return false; + } + return true; +} + + +// ---- reading /proc and the harness stores ------------------------------------ + +/// Where the view reads from. Every path below is composed from these +/// and a pid; nothing here is ever built from a client's bytes. +pub const Sources = struct { + io: Io, + /// The proc filesystem root — `/proc`, or a fixture tree under test. + proc: []const u8, + /// $HOME, for the store-slug spellings. + home: []const u8, + /// The pinned harness roots, indexed by `@intFromEnum(Kind)`; an + /// empty base means that harness is unreachable. + roots: [5][]const u8 = @splat(""), + /// The daemon's own pid: it and its ancestors are never listed, so + /// 9agents can neither show nor act on the tree it lives in. Zero + /// disables the check (fixtures have no ancestry). + self_pid: u32 = 0, + /// Seconds between the epoch and boot, from `/proc/stat`'s `btime`. + /// Zero when it could not be read; `started` is then zero too. + btime: i64 = 0, +}; + +/// USER_HZ: the unit of `/proc//stat`'s time fields. Linux reports +/// them in hundredths of a second whatever CONFIG_HZ is. +const user_hz: u64 = 100; + +fn readSmall(io: Io, path: []const u8, buf: []u8) ?[]const u8 { + const out = Io.Dir.readFile(.cwd(), io, path, buf) catch return null; + return out; +} + +fn linkOf(io: Io, path: []const u8, buf: []u8) ?[]const u8 { + const n = Io.Dir.readLinkAbsolute(io, path, buf) catch return null; + return buf[0..n]; +} + +fn statOf(io: Io, path: []const u8) ?Io.Dir.Stat { + return Io.Dir.statFile(.cwd(), io, path, .{ .follow_symlinks = true }) catch null; +} + +fn mtimeOf(io: Io, path: []const u8) i64 { + const st = statOf(io, path) orelse return 0; + return st.mtime.toSeconds(); +} + +/// `parts` joined with `/` into `buf`, or null when they do not fit. +fn join(buf: []u8, parts: []const []const u8) ?[]const u8 { + var n: usize = 0; + for (parts, 0..) |p, i| { + if (i > 0) { + if (n + 1 > buf.len) return null; + buf[n] = '/'; + n += 1; + } + if (n + p.len > buf.len) return null; + @memcpy(buf[n..][0..p.len], p); + n += p.len; + } + return buf[0..n]; +} + +/// `` of a harness, or null when it is not pinned. +fn rootOf(s: Sources, k: Kind) ?[]const u8 { + const base = s.roots[@intFromEnum(k)]; + return if (base.len == 0) null else base; +} + +/// The directory a harness keeps its session transcripts under. +fn sessionRoot(s: Sources, k: Kind, buf: []u8) ?[]const u8 { + const base = rootOf(s, k) orelse return null; + return switch (k) { + // The omp root is ~/.omp and its mirror hangs off agent/. + .omp => join(buf, &.{ base, "agent", "sessions" }), + .claude => join(buf, &.{ base, "projects" }), + .dsh => join(buf, &.{ base, "sessions" }), + // codex names each rollout after its session, so the file it + // holds open is the whole answer — no sqlite needed. + .codex => join(buf, &.{ base, "sessions" }), + // hermes keeps its sessions only in sqlite, and the only hermes + // processes that run are the gateway and the dashboard, which + // are never sessions. Nothing to resolve. + .hermes => null, + }; +} + +// ---- the table ---------------------------------------------------------------- + +pub const Slot = struct { + used: bool = false, + /// Bumped whenever the slot comes to hold a different process, so a + /// node id minted for the old one stops resolving. + gen: u8 = 0, + seen: bool = false, + live: Live = .{}, +}; + +pub const Table = struct { + slots: [max_live]Slot = @splat(.{}), + + /// The slot holding this (pid, starttime), if any. + fn slotOfPid(t: *Table, pid: u32, starttime: u64) ?*Slot { + for (&t.slots) |*sl| { + if (sl.used and sl.live.pid == pid and sl.live.starttime == starttime) return sl; + } + return null; + } + + fn freeSlot(t: *Table) ?*Slot { + for (&t.slots) |*sl| if (!sl.used) return sl; + return null; + } + + /// The live record at `index`, when its generation still matches. + pub fn at(t: *Table, index: usize, gen: u8) ?*Live { + if (index >= t.slots.len) return null; + const sl = &t.slots[index]; + if (!sl.used or sl.gen != gen) return null; + return &sl.live; + } + + pub fn indexOf(t: *Table, l: *const Live) usize { + const base = @intFromPtr(&t.slots[0]); + return (@intFromPtr(l) - @sizeOf(Slot) + @sizeOf(Slot) - @offsetOf(Slot, "live") - base) / @sizeOf(Slot); + } + + /// Rescans `/proc`. Slots keep their index and generation while the + /// same process is still there, so a handle opened before the scan + /// keeps working; a slot that comes to hold a different process + /// bumps its generation and the old node ids stop resolving. + pub fn scan(t: *Table, s: Sources) void { + for (&t.slots) |*sl| sl.seen = false; + + var anc: [64]u32 = @splat(0); + const anc_n = ancestry(s, &anc); + + var buf: [path_capacity]u8 = undefined; + if (Io.Dir.openDirAbsolute(s.io, s.proc, .{ .iterate = true })) |dir| { + defer Io.Dir.close(dir, s.io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(dir, &rb); + while (true) { + const entry = (it.next(s.io) catch break) orelse break; + const pid = std.fmt.parseInt(u32, entry.name, 10) catch continue; + var skip = false; + for (anc[0..anc_n]) |a| if (a == pid) { + skip = true; + }; + if (skip) continue; + t.consider(s, pid, &buf); + } + } else |_| {} + + for (&t.slots) |*sl| { + if (sl.used and !sl.seen) { + sl.used = false; + sl.gen +%= 1; + } + } + } + + /// Looks at one pid and takes it if it is an agent. + fn consider(t: *Table, s: Sources, pid: u32, buf: []u8) void { + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch return; + var dir_buf: [path_capacity]u8 = undefined; + const pid_dir = join(&dir_buf, &.{ s.proc, pid_text }) orelse return; + + var stat_buf: [probe_capacity]u8 = undefined; + const stat_path = join(buf, &.{ pid_dir, "stat" }) orelse return; + const stat = readSmall(s.io, stat_path, &stat_buf) orelse return; + const starttime = startTimeOf(stat) orelse return; + + var cmd_buf: [probe_capacity]u8 = undefined; + const cmd_path = join(buf, &.{ pid_dir, "cmdline" }) orelse return; + const cmdline = readSmall(s.io, cmd_path, &cmd_buf) orelse return; + const kind = harnessOf(cmdline) orelse return; + + // Already known and still the same process: keep the slot and + // everything resolved for it. + if (t.slotOfPid(pid, starttime)) |sl| { + sl.seen = true; + return; + } + + const sl = t.freeSlot() orelse return; // full: the rest simply do not list + sl.* = .{ .used = true, .gen = sl.gen +% 1, .seen = true, .live = .{} }; + const l = &sl.live; + l.kind = kind; + l.pid = pid; + l.ppid = parentOf(stat) orelse 0; + l.starttime = starttime; + l.started = if (s.btime == 0) 0 else s.btime + @as(i64, @intCast(starttime / user_hz)); + + var name_buf: [name_capacity]u8 = undefined; + l.name.set(std.fmt.bufPrint(&name_buf, "{s}-{d}", .{ kind.text(), pid }) catch kind.text()); + + const cwd_path = join(buf, &.{ pid_dir, "cwd" }) orelse return; + var cwd_buf: [cwd_capacity]u8 = undefined; + if (linkOf(s.io, cwd_path, &cwd_buf)) |cwd| l.cwd.set(cwd); + + resolveSession(l, s, pid_dir); + } +}; + +/// The daemon's own pid and every parent of it, so the tree it lives in +/// is never listed. Returns how many were written. +fn ancestry(s: Sources, out: []u32) usize { + if (s.self_pid == 0) return 0; + var n: usize = 0; + var pid = s.self_pid; + var buf: [path_capacity]u8 = undefined; + var stat_buf: [probe_capacity]u8 = undefined; + while (pid > 1 and n < out.len) { + out[n] = pid; + n += 1; + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch break; + const path = join(&buf, &.{ s.proc, pid_text, "stat" }) orelse break; + const stat = readSmall(s.io, path, &stat_buf) orelse break; + const parent = parentOf(stat) orelse break; + if (parent == 0 or parent == pid) break; + pid = parent; + } + return n; +} + +/// Asks the harness, in its own terms, which stored session this +/// process is writing. Leaves `via = .none` when it cannot tell: the +/// view never invents a session it did not find. +fn resolveSession(l: *Live, s: Sources, pid_dir: []const u8) void { + switch (l.kind) { + .claude => resolveClaude(l, s), + .omp => resolveOmp(l, s, pid_dir), + .dsh, .codex => resolveByFd(l, s, pid_dir), + // hermes has no per-session file to find. + .hermes => {}, + } +} + +/// Claude Code maintains `sessions/.json` itself, so there is +/// nothing to guess — only to check. `procStart` must match the +/// process's start time, or the record belongs to a dead predecessor +/// that happened to share the pid. +fn resolveClaude(l: *Live, s: Sources) void { + const base = rootOf(s, .claude) orelse return; + var buf: [path_capacity]u8 = undefined; + var num: [24]u8 = undefined; + const file = std.fmt.bufPrint(&num, "{d}.json", .{l.pid}) catch return; + const path = join(&buf, &.{ base, "sessions", file }) orelse return; + var text_buf: [probe_capacity]u8 = undefined; + const text = readSmall(s.io, path, &text_buf) orelse return; + + const proc_start = jsonField(text, "procStart") orelse return; + const claimed = std.fmt.parseInt(u64, proc_start, 10) catch return; + if (claimed != l.starttime) return; // a record left by a reused pid + + const sid = jsonField(text, "sessionId") orelse return; + if (sid.len == 0 or sid.len > text_capacity) return; + l.session.set(sid); + l.via = .registry; + + var slug_buf: [cwd_capacity]u8 = undefined; + const slug = slugOf(.claude, l.cwd.slice(), s.home, &slug_buf) orelse return; + var tr_buf: [path_capacity]u8 = undefined; + var file_buf: [text_capacity + 8]u8 = undefined; + const tr_name = std.fmt.bufPrint(&file_buf, "{s}.jsonl", .{sid}) catch return; + const tr = join(&tr_buf, &.{ base, "projects", slug, tr_name }) orelse return; + if (statOf(s.io, tr) != null) l.transcript.set(tr); +} + +/// omp: the transcript it holds open, else the store directory named +/// after its working directory. +fn resolveOmp(l: *Live, s: Sources, pid_dir: []const u8) void { + resolveByFd(l, s, pid_dir); + if (l.via == .none) resolveOmpByDir(l, s); + if (l.transcript.len == 0) return; + const tr = l.transcript.slice(); + l.session.set(sessionIdOf(.omp, basename(tr))); + // The subagents are sibling files in the directory named after the + // session file, one `.jsonl` each. + if (std.mem.endsWith(u8, tr, ".jsonl")) { + l.agent_dir.set(tr[0 .. tr.len - ".jsonl".len]); + } +} + +fn resolveOmpByDir(l: *Live, s: Sources) void { + var root_buf: [path_capacity]u8 = undefined; + const root = sessionRoot(s, .omp, &root_buf) orelse return; + var slug_buf: [cwd_capacity]u8 = undefined; + const slug = slugOf(.omp, l.cwd.slice(), s.home, &slug_buf) orelse return; + var dir_buf: [path_capacity]u8 = undefined; + const dir_path = join(&dir_buf, &.{ root, slug }) orelse return; + var newest_buf: [path_capacity]u8 = undefined; + const newest = newestUnder(s.io, dir_path, ".jsonl", &newest_buf) orelse return; + l.transcript.set(newest); + l.via = .dir; +} + +/// The newest entry of `dir_path` whose name ends in `suffix`, as a +/// full path written into `out`. +fn newestUnder(io: Io, dir_path: []const u8, suffix: []const u8, out: []u8) ?[]const u8 { + const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch return null; + defer Io.Dir.close(dir, io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(dir, &rb); + var best: i64 = -1; + var best_len: usize = 0; + var probe: [path_capacity]u8 = undefined; + while (true) { + const entry = (it.next(io) catch break) orelse break; + if (!std.mem.endsWith(u8, entry.name, suffix)) continue; + const path = join(&probe, &.{ dir_path, entry.name }) orelse continue; + const when = mtimeOf(io, path); + if (when <= best) continue; + if (path.len > out.len) continue; + @memcpy(out[0..path.len], path); + best = when; + best_len = path.len; + } + return if (best_len == 0) null else out[0..best_len]; +} + +/// The route that needs no knowledge of how a harness names its store: +/// whatever session file the process is holding open. The newest wins, +/// so a harness that keeps an old transcript open for reading does not +/// outvote the one it is writing. +fn resolveByFd(l: *Live, s: Sources, pid_dir: []const u8) void { + var root_buf: [path_capacity]u8 = undefined; + const root = sessionRoot(s, l.kind, &root_buf) orelse return; + + var fd_buf: [path_capacity]u8 = undefined; + const fd_dir_path = join(&fd_buf, &.{ pid_dir, "fd" }) orelse return; + const fd_dir = Io.Dir.openDirAbsolute(s.io, fd_dir_path, .{ .iterate = true }) catch return; + defer Io.Dir.close(fd_dir, s.io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(fd_dir, &rb); + + var best: i64 = -1; + var best_path: [path_capacity]u8 = undefined; + var best_len: usize = 0; + while (true) { + const entry = (it.next(s.io) catch break) orelse break; + var link_path_buf: [path_capacity]u8 = undefined; + const link_path = join(&link_path_buf, &.{ fd_dir_path, entry.name }) orelse continue; + var target_buf: [path_capacity]u8 = undefined; + const target = linkOf(s.io, link_path, &target_buf) orelse continue; + if (!std.mem.startsWith(u8, target, root)) continue; + if (target.len <= root.len or target[root.len] != '/') continue; + if (!sessionFile(l.kind, target[root.len + 1 ..])) continue; + const when = mtimeOf(s.io, target); + if (when <= best or target.len > best_path.len) continue; + @memcpy(best_path[0..target.len], target); + best = when; + best_len = target.len; + } + if (best_len == 0) return; + const found = best_path[0..best_len]; + l.transcript.set(found); + l.via = .fd; + if (l.kind == .codex) l.session.set(sessionIdOf(.codex, basename(found))); + if (l.kind == .dsh) { + // ~/.dsh/sessions//session-/session.jsonl.zstd: + // the id is the directory the transcript sits in. + const rel = found[root.len + 1 ..]; + var parts = std.mem.splitScalar(u8, rel, '/'); + _ = parts.next(); + if (parts.next()) |id| l.session.set(id); + } +} + +/// Is this path, relative to the harness's session root, one of its +/// session transcripts rather than a subagent's or something else? +fn sessionFile(kind: Kind, rel: []const u8) bool { + var depth: usize = 0; + var parts = std.mem.splitScalar(u8, rel, '/'); + while (parts.next()) |_| depth += 1; + return switch (kind) { + // /.jsonl exactly: a third component is a subagent. + .omp => depth == 2 and std.mem.endsWith(u8, rel, ".jsonl"), + .claude => depth == 2 and std.mem.endsWith(u8, rel, ".jsonl"), + .dsh => depth == 3 and std.mem.startsWith(u8, rel, "-"), + // YYYY/MM/DD/rollout--.jsonl + .codex => depth == 4 and std.mem.endsWith(u8, rel, ".jsonl") and + std.mem.startsWith(u8, basename(rel), "rollout-"), + .hermes => false, + }; +} + + +// ---- fields derived when they are read --------------------------------------- + +/// The value of `key` in a NUL-separated environment block, as +/// `/proc//environ` stores it. +pub fn envValue(environ: []const u8, key: []const u8) ?[]const u8 { + var it: Argv = .init(environ); + while (it.next()) |pair| { + if (pair.len <= key.len) continue; + if (!std.mem.startsWith(u8, pair, key)) continue; + if (pair[key.len] != '=') continue; + return pair[key.len + 1 ..]; + } + return null; +} + +/// The title key each harness writes into the head of a transcript. +fn titleKey(k: Kind) []const u8 { + return switch (k) { + .claude => "aiTitle", + else => "title", + }; +} + +/// A field from the first few records of a transcript, where the +/// harnesses put the session's title and the model it runs. Only the +/// head is read: these records are written when the session opens. +pub fn headField(io: Io, kind: Kind, transcript: []const u8, which: enum { title, model }, out: []u8) ?[]const u8 { + if (transcript.len == 0) return null; + var head: [probe_capacity]u8 = undefined; + const text = readSmall(io, transcript, &head) orelse return null; + const key = switch (which) { + .title => titleKey(kind), + .model => "model", + }; + const v = jsonField(text, key) orelse return null; + if (v.len == 0 or v.len > out.len) return null; + @memcpy(out[0..v.len], v); + return out[0..v.len]; +} + +/// The zmx session a process runs inside, from its environment. +pub fn zmxOf(s: Sources, pid: u32, out: []u8) ?[]const u8 { + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{pid}) catch return null; + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ s.proc, pid_text, "environ" }) orelse return null; + var env_buf: [probe_capacity]u8 = undefined; + const env = readSmall(s.io, path, &env_buf) orelse return null; + const v = envValue(env, "ZMX_SESSION") orelse return null; + if (v.len == 0 or v.len > out.len) return null; + @memcpy(out[0..v.len], v); + return out[0..v.len]; +} + +/// A field of Claude Code's own session record, which is the only +/// harness that publishes a name and a status. +pub fn registryField(s: Sources, l: *const Live, key: []const u8, out: []u8) ?[]const u8 { + if (l.kind != .claude) return null; + const base = rootOf(s, .claude) orelse return null; + var num: [24]u8 = undefined; + const file = std.fmt.bufPrint(&num, "{d}.json", .{l.pid}) catch return null; + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ base, "sessions", file }) orelse return null; + var text_buf: [probe_capacity]u8 = undefined; + const text = readSmall(s.io, path, &text_buf) orelse return null; + // The record must still describe this process, not a reused pid. + const proc_start = jsonField(text, "procStart") orelse return null; + const claimed = std.fmt.parseInt(u64, proc_start, 10) catch return null; + if (claimed != l.starttime) return null; + const v = jsonField(text, key) orelse return null; + if (v.len == 0 or v.len > out.len) return null; + @memcpy(out[0..v.len], v); + return out[0..v.len]; +} + +/// Is this process still the one the slot remembers? A pid alone is not +/// an identity: the kernel reuses them. +pub fn stillAlive(s: Sources, l: *const Live) bool { + var num: [24]u8 = undefined; + const pid_text = std.fmt.bufPrint(&num, "{d}", .{l.pid}) catch return false; + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ s.proc, pid_text, "stat" }) orelse return false; + var stat_buf: [probe_capacity]u8 = undefined; + const stat = readSmall(s.io, path, &stat_buf) orelse return false; + const now = startTimeOf(stat) orelse return false; + return now == l.starttime; +} + +/// Seconds between the epoch and boot, from `/proc/stat`'s `btime` +/// line. Zero when it cannot be read. +pub fn bootTime(io: Io, proc: []const u8) i64 { + var path_buf: [path_capacity]u8 = undefined; + const path = join(&path_buf, &.{ proc, "stat" }) orelse return 0; + var buf: [probe_capacity]u8 = undefined; + const text = readSmall(io, path, &buf) orelse return 0; + var lines = std.mem.splitScalar(u8, text, '\n'); + while (lines.next()) |line| { + if (!std.mem.startsWith(u8, line, "btime ")) continue; + return std.fmt.parseInt(i64, std.mem.trim(u8, line["btime ".len..], " \r"), 10) catch 0; + } + return 0; +} + + +// ---- subagents ---------------------------------------------------------------- + +/// Longest subagent name answered. +pub const agent_name_capacity: usize = 64; + +/// The subagents of one session: omp writes each as `.jsonl` in a +/// directory named after the session file. Sorted, so a listing that +/// spans several reads keeps its cursor. +pub const Agents = struct { + names: [max_agents][agent_name_capacity]u8 = undefined, + lens: [max_agents]u8 = @splat(0), + count: usize = 0, + + pub fn name(a: *const Agents, i: usize) []const u8 { + if (i >= a.count) return ""; + return a.names[i][0..a.lens[i]]; + } + + pub fn indexOf(a: *const Agents, want: []const u8) ?usize { + for (0..a.count) |i| if (std.mem.eql(u8, a.name(i), want)) return i; + return null; + } +}; + +pub fn agentsOf(io: Io, l: *const Live, out: *Agents) void { + out.count = 0; + const dir_path = l.agent_dir.slice(); + if (dir_path.len == 0) return; + const dir = Io.Dir.openDirAbsolute(io, dir_path, .{ .iterate = true }) catch return; + defer Io.Dir.close(dir, io); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(dir, &rb); + while (out.count < max_agents) { + const entry = (it.next(io) catch break) orelse break; + if (!std.mem.endsWith(u8, entry.name, ".jsonl")) continue; + const stem = entry.name[0 .. entry.name.len - ".jsonl".len]; + if (stem.len == 0 or stem.len > agent_name_capacity) continue; + @memcpy(out.names[out.count][0..stem.len], stem); + out.lens[out.count] = @intCast(stem.len); + out.count += 1; + } + // Insertion sort: the list is at most `max_agents` long and already + // nearly ordered, and this keeps the struct allocation-free. + var i: usize = 1; + while (i < out.count) : (i += 1) { + var j = i; + while (j > 0 and std.mem.order(u8, out.name(j), out.name(j - 1)) == .lt) : (j -= 1) { + std.mem.swap([agent_name_capacity]u8, &out.names[j], &out.names[j - 1]); + std.mem.swap(u8, &out.lens[j], &out.lens[j - 1]); + } + } +} + +/// The transcript path of subagent `i`, written into `out`. +pub fn agentPath(l: *const Live, a: *const Agents, i: usize, out: []u8) ?[]const u8 { + if (i >= a.count) return null; + var name_buf: [agent_name_capacity + 8]u8 = undefined; + const file = std.fmt.bufPrint(&name_buf, "{s}.jsonl", .{a.name(i)}) catch return null; + return join(out, &.{ l.agent_dir.slice(), file }); +} + +// ---- unit tests --------------------------------------------------------------- + +const testing = std.testing; + +/// A `/proc//cmdline`: NUL-separated argv, written out literally so +/// the tests read the way the kernel stores it. +fn cmd(comptime words: []const []const u8) []const u8 { + comptime var out: []const u8 = ""; + inline for (words) |w| out = out ++ w ++ "\x00"; + return out; +} + +test "active: a command line names its harness, or nothing" { + try testing.expectEqual(Kind.claude, harnessOf(cmd(&.{ "claude", "--dangerously-skip-permissions" })).?); + try testing.expectEqual(Kind.omp, harnessOf(cmd(&.{"omp"})).?); + try testing.expectEqual(Kind.codex, harnessOf(cmd(&.{ "/usr/bin/codex", "resume" })).?); + try testing.expectEqual(Kind.dsh, harnessOf(cmd(&.{"/home/goblin/.local/bin/dsh"})).?); + // hermes runs as a python module, named by a later word. + try testing.expectEqual(Kind.hermes, harnessOf(cmd(&.{ "/venv/bin/python", "-m", "hermes_cli.main" })).?); + // Not a harness at all. + try testing.expect(harnessOf(cmd(&.{ "bash", "-c", "claude" })) == null); + try testing.expect(harnessOf("") == null); + try testing.expect(harnessOf("\x00\x00") == null); +} + +test "active: a daemon or helper wearing a harness name is never a session" { + // The live machine's own gateway and dashboard, which must not list. + try testing.expect(harnessOf(cmd(&.{ "python3", "-m", "hermes_cli.main", "gateway", "run" })) == null); + try testing.expect(harnessOf(cmd(&.{ "python3", "/x/hermes-dashboard", "dashboard" })) == null); + try testing.expect(harnessOf(cmd(&.{ "claude", "mcp-server" })) == null); + try testing.expect(harnessOf(cmd(&.{ "omp", "__omp_worker_daemon_broker" })) == null); + try testing.expect(harnessOf(cmd(&.{ "codex", "app-server" })) == null); + // A directory that merely contains the word is still a session: the + // exclusion matches a whole argument, never a substring. + try testing.expectEqual(Kind.claude, harnessOf(cmd(&.{ "claude", "--cwd", "/home/goblin/serve-me" })).?); +} + +test "active: starttime is counted from the last ')' in a stat line" { + // Fields: pid (comm) state ppid ... starttime is field 22. + var line: [512]u8 = undefined; + var w = Io.Writer.fixed(&line); + try w.print("345104 (claude) S 344980", .{}); + for (5..22) |i| try w.print(" {d}", .{i * 10}); + try w.print(" 1625063 rest of the line", .{}); + try testing.expectEqual(@as(u64, 1625063), startTimeOf(w.buffered()).?); + + // A process whose name holds spaces and parentheses — the reason + // the fields are never counted from the start of the line. + var nasty: [512]u8 = undefined; + var nw = Io.Writer.fixed(&nasty); + try nw.print("42 (we i)rd (name) ) R 1", .{}); + for (5..22) |i| try nw.print(" {d}", .{i}); + try nw.print(" 999 x", .{}); + try testing.expectEqual(@as(u64, 999), startTimeOf(nw.buffered()).?); + + try testing.expect(startTimeOf("no parens here") == null); + try testing.expect(startTimeOf("1 (short) S 2 3") == null); +} + +test "active: json fields come out of a harness's own session record" { + // The shape Claude Code writes to ~/.claude/sessions/.json. + const rec = + \\{ + \\ "pid": 345104, + \\ "sessionId": "1f9a74f7-b04f-4092-8b98-df11316e03c0", + \\ "cwd": "/home/goblin", + \\ "version": "2.1.278", + \\ "peerFeatures": ["notify_idle", "artifact_yield"], + \\ "name": "goblin-e5", + \\ "status": "busy", + \\ "procStart": "1625063" + \\} + ; + try testing.expectEqualStrings("345104", jsonField(rec, "pid").?); + try testing.expectEqualStrings("1f9a74f7-b04f-4092-8b98-df11316e03c0", jsonField(rec, "sessionId").?); + try testing.expectEqualStrings("/home/goblin", jsonField(rec, "cwd").?); + try testing.expectEqualStrings("goblin-e5", jsonField(rec, "name").?); + try testing.expectEqualStrings("busy", jsonField(rec, "status").?); + try testing.expectEqualStrings("1625063", jsonField(rec, "procStart").?); + // An array or object value is skipped, not half-parsed. + try testing.expect(jsonField(rec, "peerFeatures") == null); + try testing.expect(jsonField(rec, "absent") == null); + // A key that is a prefix of another must not match it. + try testing.expect(jsonField("{\"sessionIdent\":\"x\"}", "sessionId") == null); +} + +test "active: session ids and store slugs follow each harness's spelling" { + try testing.expectEqualStrings( + "01a0c46a-7a06-7676-aef1-add9d9340c83", + sessionIdOf(.omp, "2026-09-21T14-41-47-526Z_01a0c46a-7a06-7676-aef1-add9d9340c83.jsonl"), + ); + try testing.expectEqualStrings( + "42898558-2fa0-41f2-a2c6-1357e5cf6836", + sessionIdOf(.claude, "42898558-2fa0-41f2-a2c6-1357e5cf6836.jsonl"), + ); + // codex's timestamp holds dashes too, so the id is the last 36 + // bytes and never what splitting on `-` would give. + try testing.expectEqualStrings( + "01a0bcd2-5653-7cc0-aa36-676f6467a72a", + sessionIdOf(.codex, "rollout-2026-09-20T00-18-16-01a0bcd2-5653-7cc0-aa36-676f6467a72a.jsonl"), + ); + + var buf: [256]u8 = undefined; + // omp strips $HOME and keeps everything but the slashes. + try testing.expectEqualStrings( + "-00-projects-0x4200.cafe-cloud9", + slugOf(.omp, "/home/goblin/00-projects/0x4200.cafe/cloud9", "/home/goblin", &buf).?, + ); + // Claude Code dashes every byte that is not a letter or a digit. + try testing.expectEqualStrings( + "-home-goblin-00-projects-0x4200-cafe-cloud9", + slugOf(.claude, "/home/goblin/00-projects/0x4200.cafe/cloud9", "/home/goblin", &buf).?, + ); + // A path longer than the buffer is refused, never truncated into a + // slug that would name the wrong store. + var tiny: [4]u8 = undefined; + try testing.expect(slugOf(.omp, "/home/goblin/somewhere", "/home/goblin", &tiny) == null); +} + +test "active: a zmx session name is checked before it is ever used" { + try testing.expect(legalZmxName("harness")); + try testing.expect(legalZmxName("claude-cloud9.2_x")); + try testing.expect(!legalZmxName("")); + try testing.expect(!legalZmxName("has space")); + try testing.expect(!legalZmxName("../escape")); + try testing.expect(!legalZmxName("semi;colon")); + try testing.expect(!legalZmxName("nul\x00byte")); + try testing.expect(!legalZmxName("x" ** 65)); +} + +test "active: an environment block answers by key, not by prefix" { + const env = "PATH=/usr/bin\x00ZMX_SESSION=harness\x00HOME=/home/goblin\x00"; + try testing.expectEqualStrings("harness", envValue(env, "ZMX_SESSION").?); + try testing.expectEqualStrings("/home/goblin", envValue(env, "HOME").?); + try testing.expect(envValue(env, "ZMX") == null); // a prefix is not the key + try testing.expect(envValue(env, "NOPE") == null); + try testing.expect(envValue("", "HOME") == null); + // An entry with no value is a value of zero length, not a miss. + try testing.expectEqualStrings("", envValue("EMPTY=\x00", "EMPTY").?); +} diff --git a/9agents/src/main.zig b/9agents/src/main.zig new file mode 100644 index 0000000..2ac9078 --- /dev/null +++ b/9agents/src/main.zig @@ -0,0 +1,382 @@ +//! 9agents: the coding-agents fs daemon — a long-running 9P2000 server that +//! mirrors every AI-agent harness's state (Claude Code, Codex, omp, +//! hermes, dsh + their skills) as one read-only, fresh-from-disk tree. +//! See docs/DESIGN.md for the tree contract and src/tree.zig for the tree. +//! +//! By default it posts itself under the name `agents` with +//! `serve.Runner.listenPosted`, so it appears at $XDG_RUNTIME_DIR/9p/agents +//! and every interactive fish (self-wrapped in `9ns --mntgen`) sees it at +//! /mnt/9p/agents with zero configuration. `--unix`, `--tcp` and `--fd` +//! are the other listen forms; `--no-post` serves without posting; the +//! five harness roots default to $HOME/. and `--root NAME=PATH` +//! pins any of them elsewhere (the tests use fake homes). +//! +//! v1 has no daemonization: it never forks or detaches, so it is run under +//! something that supervises it — a systemd user service with +//! `Restart=always` (see 9agents.service), or a zmx session: +//! +//! zmx run agents -d 9agents +//! +//! SIGTERM or SIGINT stops it cleanly (a posted name is unposted). +const std = @import("std"); +const builtin = @import("builtin"); +const cloud9 = @import("cloud9"); +const tree = @import("tree.zig"); + +const fs = cloud9.fs; +const serve = cloud9.serve; +const post = cloud9.post; +const transport = cloud9.transport; +const Io = std.Io; +const linux = std.os.linux; + +const usage_text = + \\usage: 9agents [--unix PATH | --tcp IP:PORT | --fd N] [--no-post] + \\ [--name NAME] [--root NAME=PATH]... [--proc DIR] + \\ [--allow-move] [--zmx PATH] + \\ + \\A read-only, fresh-from-disk 9P2000 view of every harness's state: + \\ /README the tree, explained in place + \\ /pid /uptime daemon facts + \\ /claude/{projects,history,skills} + \\ /codex/{sessions,session-index,history} + \\ /omp /hermes /dsh full mirrors, raw + \\ /skills/{claude,codex,omp} the union skills view + \\ /active/// what is running right now + \\ + \\By default the daemon posts itself under the name `agents`, so it is + \\dialable at $XDG_RUNTIME_DIR/9p/agents and mountable by 9ns --mntgen + \\at /mnt/9p/agents. --no-post skips posting; --unix/--tcp add plain + \\listeners beside the post; --fd N serves one 9P session over the + \\connected stream on descriptor N and posts nothing. + \\--root NAME=PATH pins one harness root (NAME: claude, codex, omp, + \\hermes, dsh) somewhere other than $HOME/.; repeatable. + \\--proc DIR reads the process tree somewhere other than /proc (for + \\tests); --proc "" leaves /active out of the tree entirely. + \\--allow-move lets a write to /active///zmx move that agent + \\into a zmx session of that name: the daemon kills it and re-execs + \\the harness under zmx with its session resumed. Off by default, + \\because it is the one place the tree is not read-only, and anything + \\that can mount it can then kill an agent. --zmx PATH names the + \\binary a move runs (default: zmx, found on $PATH). + \\ + \\Run it in a zmx session, the zmx way: `zmx run agents -d 9agents`. + \\ +; + +const max_connections = 16; +const Runner = serve.Runner(tree.Harness, tree.opts, .{ + .msize = tree.msize, + .connections = max_connections, + .listeners = 2, // the posted name plus one of --unix/--tcp +}); + +// Static memory, the 9proc-demo way: the harness and the runner are large +// (the path table, the per-connection buffers) and live in .bss. +var harness_mem: tree.Harness = undefined; +var runner_mem: Runner = undefined; +var stop_requested = std.atomic.Value(bool).init(false); + +fn onSignal(sig: linux.SIG) callconv(.c) void { + _ = sig; + stop_requested.store(true, .release); +} + +fn installSignalHandlers() void { + const act = linux.Sigaction{ + .handler = .{ .handler = onSignal }, + .mask = @splat(0), + .flags = 0, + }; + _ = linux.sigaction(.TERM, &act, null); + _ = linux.sigaction(.INT, &act, null); + const ign = linux.Sigaction{ + .handler = .{ .handler = linux.SIG.IGN }, + .mask = @splat(0), + .flags = 0, + }; + _ = linux.sigaction(.PIPE, &ign, null); +} + +// ---- the serve handler --------------------------------------------------------- + +fn serveReq(ctx: ?*anyopaque, conn: *Runner.Conn, req: fs.Req) void { + const h: *tree.Harness = @ptrCast(@alignCast(ctx.?)); + // The answer's bytes point into the harness's shared buffers, so the + // mutex spans handle and reply: the engine copies them into the + // connection's output before the lock goes. + h.mutex.lockUncancelable(h.io); + const a = tree.handle(h, req); + conn.reply(&a.reply, a.bytes); + h.mutex.unlock(h.io); +} + +/// `name` found on `path_env`, as an absolute path in the arena. +fn onPath(io: Io, arena: std.mem.Allocator, path_env: []const u8, name: []const u8) ![]const u8 { + var it = std.mem.splitScalar(u8, path_env, ':'); + while (it.next()) |dir_path| { + if (dir_path.len == 0) continue; + const candidate = try std.fmt.allocPrint(arena, "{s}/{s}", .{ dir_path, name }); + const st = Io.Dir.statFile(.cwd(), io, candidate, .{ .follow_symlinks = true }) catch continue; + if (st.kind != .file) continue; + return candidate; + } + return error.NotFound; +} + +// ---- the CLI -------------------------------------------------------------------- + +const Mode = enum { posted, unix, tcp, fd }; + +pub fn main(init: std.process.Init) !void { + run(init) catch |e| switch (e) { + // Both are reported with their own line above; a returned error + // would bury it under a Debug-build stack trace. + error.Usage, error.AlreadyPosted => std.process.exit(1), + else => return e, + }; +} + +fn run(init: std.process.Init) !void { + const io = init.io; + const arena = init.arena.allocator(); + const args = try init.minimal.args.toSlice(arena); + const envp: post.Env = init.minimal.environ.block.slice.ptr; + + var mode: Mode = .posted; + var unix_path: []const u8 = ""; + var tcp_addr: []const u8 = ""; + var fd_no: i32 = -1; + var post_name: []const u8 = "agents"; + var no_post = false; + var base_overrides: [5]?[]const u8 = @splat(null); + var proc_root: []const u8 = "/proc"; + var zmx_path: []const u8 = "zmx"; + var allow_move = false; + + var i: usize = 1; + while (i < args.len) : (i += 1) { + const a = args[i]; + if (std.mem.eql(u8, a, "--no-post")) { + no_post = true; + } else if (std.mem.eql(u8, a, "--allow-move")) { + allow_move = true; + } else if (std.mem.eql(u8, a, "--unix") or std.mem.eql(u8, a, "--tcp") or + std.mem.eql(u8, a, "--fd") or std.mem.eql(u8, a, "--name") or + std.mem.eql(u8, a, "--proc") or std.mem.eql(u8, a, "--zmx") or + std.mem.eql(u8, a, "--root")) + { + i += 1; + if (i >= args.len) { + std.debug.print("9agents: {s} needs an argument\n{s}", .{ a, usage_text }); + return error.Usage; + } + const v = args[i]; + if (std.mem.eql(u8, a, "--unix")) { + mode = .unix; + unix_path = v; + } else if (std.mem.eql(u8, a, "--tcp")) { + mode = .tcp; + tcp_addr = v; + } else if (std.mem.eql(u8, a, "--fd")) { + mode = .fd; + fd_no = std.fmt.parseInt(i32, v, 10) catch { + std.debug.print("9agents: --fd: not a number: {s}\n", .{v}); + return error.Usage; + }; + } else if (std.mem.eql(u8, a, "--proc")) { + proc_root = v; + } else if (std.mem.eql(u8, a, "--zmx")) { + zmx_path = v; + } else if (std.mem.eql(u8, a, "--name")) { + post_name = v; + if (!post.legalName(post_name)) { + std.debug.print("9agents: illegal post name: {s}\n", .{v}); + return error.Usage; + } + } else { + const eq = std.mem.indexOfScalar(u8, v, '=') orelse { + std.debug.print("9agents: --root NAME=PATH: {s}\n{s}", .{ v, usage_text }); + return error.Usage; + }; + const name = v[0..eq]; + const root = std.meta.stringToEnum(tree.Root, name) orelse { + std.debug.print("9agents: unknown root {s} (claude, codex, omp, hermes, dsh)\n", .{name}); + return error.Usage; + }; + base_overrides[@intFromEnum(root)] = v[eq + 1 ..]; + } + } else if (std.mem.eql(u8, a, "--help") or std.mem.eql(u8, a, "-h")) { + std.debug.print("{s}", .{usage_text}); + return; + } else { + std.debug.print("9agents: unknown argument {s}\n{s}", .{ a, usage_text }); + return error.Usage; + } + } + + // The five pinned roots: $HOME/. unless overridden. + const home = post.getenv(envp, "HOME"); + var bases: [5][]const u8 = @splat(""); + inline for (0..5) |r| { + if (base_overrides[r]) |path| { + bases[r] = path; + } else if (home) |hm| { + bases[r] = std.fmt.bufPrint(&root_bufs[r], "{s}/.{s}", .{ hm, root_dirs[r] }) catch return error.NameTooLong; + } + } + var any = false; + for (bases) |b| any = any or b.len != 0; + if (!any) { + std.debug.print("9agents: no harness roots (set $HOME or pass --root NAME=PATH)\n", .{}); + return error.Usage; + } + + // A move execs zmx by absolute path: the daemon resolves it once, + // at startup, so nothing about $PATH matters when the write lands. + const zmx_abs = if (std.mem.indexOfScalar(u8, zmx_path, '/') != null) + zmx_path + else + onPath(io, arena, post.getenv(envp, "PATH") orelse "", zmx_path) catch zmx_path; + if (allow_move and std.mem.indexOfScalar(u8, zmx_abs, '/') == null) { + std.debug.print("9agents: --allow-move needs zmx on $PATH (or --zmx PATH)\n", .{}); + return error.Usage; + } + + harness_mem.init(.{ + .io = io, + .pid = @intCast(linux.getpid()), + .bases = bases, + .proc = proc_root, + .home = home orelse "", + .zmx = zmx_abs, + .runtime = post.getenv(envp, "XDG_RUNTIME_DIR") orelse "", + .allow_move = allow_move, + .envp = envp, + }); + if (allow_move) { + std.debug.print("9agents: moves allowed — a write to /active///zmx re-execs that agent under {s}\n", .{zmx_abs}); + } + if (no_post and mode == .posted) { + std.debug.print("9agents: --no-post needs a listen form (--unix, --tcp or --fd)\n{s}", .{usage_text}); + return error.Usage; + } + + installSignalHandlers(); + + switch (mode) { + .fd => { + std.debug.print("9agents: serving one session on fd {d}\n", .{fd_no}); + try serveFd(io, fd_no); + return; + }, + .posted, .unix, .tcp => {}, + } + + runner_mem.init(.{ .io = io, .root = tree.root, .handler = .{ .ctx = &harness_mem, .serve = serveReq } }); + defer runner_mem.stop(); + + var posted_something = false; + if (!no_post) { + runner_mem.listenPosted(envp, post_name, 16) catch |err| { + if (err == error.AlreadyPosted) { + std.debug.print("9agents: the name `{s}` is already posted by a live server\n", .{post_name}); + return err; + } + std.debug.print("9agents: cannot post as {s}: {t}\n", .{ post_name, err }); + return err; + }; + posted_something = true; + var pbuf: [transport.sun_path_len]u8 = undefined; + const path = post.registryPath(envp, post_name, &pbuf) catch ""; + std.debug.print("9agents: posted as {s} at {s}\n", .{ post_name, path }); + } + switch (mode) { + .unix => { + _ = try runner_mem.listen(.{ .unix = try arena.dupeZ(u8, unix_path) }, 16); + std.debug.print("9agents: listening on {s}\n", .{unix_path}); + }, + .tcp => { + const addr = Io.net.IpAddress.parseLiteral(tcp_addr) catch { + std.debug.print("9agents: bad --tcp address: {s}\n", .{tcp_addr}); + return error.Usage; + }; + const bound = try runner_mem.listen(.{ .tcp = addr }, 16); + std.debug.print("9agents: listening on tcp!{f}\n", .{bound}); + }, + else => {}, + } + var pinned: u32 = 0; + for (bases) |b| { + if (b.len != 0) pinned += 1; + } + std.debug.print("9agents: serving ({d} roots pinned, {d} bytes of tables)\n", .{ pinned, @sizeOf(tree.Harness) }); + + // The runner's tasks drive themselves; the main thread only waits for + // a stop signal (SIGTERM/SIGINT) and then unposts via stop(). + while (!stop_requested.load(.acquire)) { + io.sleep(.fromMilliseconds(250), .awake) catch break; + } + std.debug.print("9agents: stopping\n", .{}); +} + +const root_dirs = [5][]const u8{ "claude", "codex", "omp", "hermes", "dsh" }; +var root_bufs: [5][tree.base_capacity]u8 = @splat(@splat(0)); + +// ---- --fd N: one 9P session over a connected stream ----------------------------- + +/// Serves exactly one client on the connected stream of descriptor `n` +/// (socket activation, `9ns --spawn`-style handoffs), driving the engine +/// directly, the freestanding way. Returns when the client hangs up. +fn serveFd(io: Io, n: i32) !void { + const stream: Io.net.Stream = .{ .socket = .{ .handle = n, .address = .{ .ip4 = .loopback(0) } } }; + const Engine = fs.Server(tree.Harness, tree.opts); + var in: [tree.msize]u8 = undefined; + var out: [2 * tree.msize]u8 = undefined; + var rbuf: [tree.msize]u8 = undefined; + var wbuf: [2 * tree.msize]u8 = undefined; + var stage: [tree.msize]u8 = undefined; + var engine = Engine.init(.{ .in = &in, .out = &out, .root = tree.root }); + var reader = stream.reader(io, &rbuf); + var writer = stream.writer(io, &wbuf); + while (true) { + const frame = transport.readFrame(&reader.interface, &stage, tree.msize) catch return; + var off: usize = 0; + while (off < frame.len) { + off += engine.push(frame[off..]); + while (true) { + const req = engine.next() orelse break; + const a = answer(req); + engine.reply(&a.reply, a.bytes); + } + const pending = engine.output(); + writer.interface.writeAll(pending) catch return; + engine.wrote(pending.len); + if (engine.protocol.dead) return; + } + writer.interface.flush() catch return; + } +} + +/// One request for the --fd loop: same locking discipline as the runner's +/// handler (the reply bytes live in the harness's shared buffers). +fn answer(req: fs.Req) tree.Answer { + harness_mem.mutex.lockUncancelable(harness_mem.io); + defer harness_mem.mutex.unlock(harness_mem.io); + return tree.handle(&harness_mem, req); +} + +test { + _ = @import("tree.zig"); // the tree's unit tests (exclusions, ids, ...) +} + +test "main: the root override parser pins each named root" { + // Parsing is inline in run(); the mapping itself is what the tests + // rely on, and it is exercised end to end in test/e2e.sh. + _ = std.meta.stringToEnum(tree.Root, "claude").?; + _ = std.meta.stringToEnum(tree.Root, "codex").?; + _ = std.meta.stringToEnum(tree.Root, "omp").?; + _ = std.meta.stringToEnum(tree.Root, "hermes").?; + _ = std.meta.stringToEnum(tree.Root, "dsh").?; + try std.testing.expect(std.meta.stringToEnum(tree.Root, "zmx") == null); +} diff --git a/9agents/src/tree.zig b/9agents/src/tree.zig new file mode 100644 index 0000000..ed0699c --- /dev/null +++ b/9agents/src/tree.zig @@ -0,0 +1,2413 @@ +//! The harness file tree served over 9P2000: a unified, read-only, +//! fresh-from-disk view of every AI-agent harness's state on the machine. +//! +//! /README what the tree is and how to read it +//! /pid /uptime daemon facts, one line each +//! /claude/ projects/ (mirror of ~/.claude/projects), +//! history (~/.claude/history.jsonl), +//! skills/ (mirror of ~/.claude/skills) +//! /codex/ sessions/ (~/.codex/sessions), +//! session-index (~/.codex/session_index.jsonl), +//! history (~/.codex/history.jsonl) +//! /omp/ mirror of ~/.omp/agent, raw blobs +//! /hermes/ mirror of ~/.hermes, raw +//! /dsh/ mirror of ~/.dsh +//! /skills/{claude,codex,omp}/ the same skills trees the harnesses own +//! +//! Everything below the named mount points is a lazy mirror: a lookup, +//! getattr or readdir walks the real filesystem at request time, so a +//! session transcript grows as its harness writes it and a new session +//! appears as soon as its file lands. No cache, no invalidation. +//! +//! Security boundary — the exclusion rule is absolute and unit-tested: +//! no name that looks like a credential, key, token or auth store is ever +//! answered, at any depth (`excluded`); symlinks are never served (they +//! are an escape hatch around the pinned roots). Every path is resolved +//! from its pinned root one component at a time with `O_NOFOLLOW` +//! (`openIn`), so no name below a root — swapped mid-session or not — +//! can point the daemon at a file outside it. The daemon must never +//! become a credential reader for anything that mounts it. Writes, +//! creates and setattrs answer EPERM. +//! +//! This module is the backend of `cloud9.fs.Server` (main.zig hands it to +//! `serve.Runner`). It declares `features = .{ .references = true }`: every +//! lookup result is a reference, so table entries below (the mirrored +//! files) are refcounted and freed when the last fid lets go. Node ids +//! pack (kind, root index, serial) in a u64, zmx-style: a mirrored file's +//! serial is its table slot and generation, so two walks of the same file +//! answer the same qid path while it is held. +const std = @import("std"); +const cloud9 = @import("cloud9"); +const fs = cloud9.fs; +const E = fs.E; +const Io = std.Io; +const linux = std.os.linux; +pub const active = @import("active.zig"); + +// ---- comptime bounds --------------------------------------------------------- + +/// Longest relative path a mirrored file may carry (bytes below a root). +pub const rel_capacity: usize = 640; +/// Mirrored files remembered at once; each holds one reference per fid. +pub const path_capacity: usize = 4096; +comptime { + if (path_capacity > 1 << 12) @compileError("path table slots must fit the 12 serial bits"); +} +/// Largest frame the daemon negotiates; read and readdir staging buffers +/// are this big, so a single Rread can carry one full frame of bytes. +pub const msize: u32 = 32 * 1024; +/// Names one directory listing may stage (sorted for cursor stability). +pub const list_capacity: usize = 1024; +/// Bytes of name storage one listing may use (255 per name, packed). +pub const list_name_bytes: usize = 128 * 1024; +/// Longest base path a pinned root may carry. +pub const base_capacity: usize = 512; + +pub const opts: fs.Options = .{ + .fid_capacity = 512, + .slot_capacity = 8, + .name_capacity = 255, + .username_capacity = 28, + .fid_index = true, +}; + +// ---- the tree skeleton ------------------------------------------------------ + +pub const Root = enum(u8) { claude, codex, omp, hermes, dsh }; + +/// Static nodes: the facts, the harness dirs and the union skills dir. +pub const Top = enum(u8) { + root = 1, + pid, + uptime, + claude, + codex, + omp, + hermes, + dsh, + skills, + active, + readme, + + /// Directory entries of the root, in listing order. + pub const listed = [_]Top{ .readme, .pid, .uptime, .claude, .codex, .omp, .hermes, .dsh, .skills, .active }; + + pub fn fileName(t: Top) []const u8 { + // The only name that is not its tag: a README announces itself in + // the listing, and lowercase would bury it among the harness dirs. + return if (t == .readme) "README" else @tagName(t); + } + + pub fn dir(t: Top) bool { + return switch (t) { + .root, .claude, .codex, .omp, .hermes, .dsh, .skills, .active => true, + .pid, .uptime, .readme => false, + }; + } + + fn parent(t: Top) Top { + _ = t; + return .root; // the root is its own parent; every top hangs off it + } + + /// The mirror a harness dir serves: its root and the relative path + /// below it. The claude and codex dirs are *virtual* (their children + /// are named mounts); omp, hermes and dsh mirror one subtree each. + fn mirror(t: Top) ?Mount { + return switch (t) { + .omp => .{ .root = .omp, .rel = "agent" }, + .hermes => .{ .root = .hermes, .rel = "" }, + .dsh => .{ .root = .dsh, .rel = "" }, + else => null, + }; + } +}; + +/// Where a mirrored subtree sits: `rel` below the root's pinned base path. +pub const Mount = struct { + root: Root, + rel: []const u8, +}; + +const NamedMount = struct { + /// The name as it appears in its virtual directory. + name: []const u8, + owner: Top, + mount: Mount, +}; + +/// The children of the virtual directories (claude, codex, skills). The +/// same target may appear twice (/claude/skills and /skills/claude): a +/// lookup of either hands out the same table entry, so both paths share +/// qid paths and identity. +const named_mounts = [_]NamedMount{ + .{ .name = "projects", .owner = .claude, .mount = .{ .root = .claude, .rel = "projects" } }, + .{ .name = "history", .owner = .claude, .mount = .{ .root = .claude, .rel = "history.jsonl" } }, + .{ .name = "skills", .owner = .claude, .mount = .{ .root = .claude, .rel = "skills" } }, + .{ .name = "sessions", .owner = .codex, .mount = .{ .root = .codex, .rel = "sessions" } }, + .{ .name = "session-index", .owner = .codex, .mount = .{ .root = .codex, .rel = "session_index.jsonl" } }, + .{ .name = "history", .owner = .codex, .mount = .{ .root = .codex, .rel = "history.jsonl" } }, + .{ .name = "claude", .owner = .skills, .mount = .{ .root = .claude, .rel = "skills" } }, + .{ .name = "codex", .owner = .skills, .mount = .{ .root = .codex, .rel = "skills" } }, + .{ .name = "omp", .owner = .skills, .mount = .{ .root = .omp, .rel = "skills" } }, +}; + +fn mountNamed(owner: Top, name: []const u8) ?Mount { + for (named_mounts) |m| { + if (m.owner == owner and std.mem.eql(u8, m.name, name)) return m.mount; + } + return null; +} + +/// The Top that owns the mount whose target is exactly (root, rel) — the +/// canonical parent a ".." walk answers. Falls back to the root's harness +/// dir for targets no named mount covers (deep paths, mirrors). +fn mountOwner(r: Root, rel_path: []const u8) Top { + for (named_mounts) |m| { + if (m.mount.root == r and std.mem.eql(u8, m.mount.rel, rel_path)) return m.owner; + } + return switch (r) { + .claude => .claude, + .codex => .codex, + .omp => .omp, + .hermes => .hermes, + .dsh => .dsh, + }; +} + +// ---- node ids and the path table --------------------------------------------- + +pub const Kind = enum(u8) { top = 0, path = 1, active = 2 }; + +comptime { + // The derived view indexes the same pinned roots the mirror does. + for (std.enums.values(Root)) |r| { + if (!std.mem.eql(u8, @tagName(r), @tagName(@as(active.Kind, @enumFromInt(@intFromEnum(r)))))) + @compileError("tree.Root and active.Kind must agree, index by index"); + } +} + +/// A node under `/active`. Everything but a harness directory names a +/// slot of the live table and the generation it had when the id was +/// minted, so an id outlives its process only long enough to answer +/// ENOENT. +pub const AFile = enum(u8) { + /// `/active/` + harness_dir, + /// `/active//` + entry, + pid, + ppid, + started, + cwd, + name, + title, + session, + model, + via, + zmx, + status, + transcript, + /// `/active///agents` + agents, + /// `/active///agents/` + agent, + agent_model, + agent_transcript, + + /// The fields of one entry, in listing order. A field with no value + /// is not listed and does not resolve. + pub const entry_files = [_]AFile{ + .pid, .ppid, .started, .cwd, .name, .title, + .session, .model, .via, .zmx, .status, .transcript, + }; + pub const agent_files = [_]AFile{ .agent_model, .agent_transcript }; + + pub fn fileName(f: AFile) []const u8 { + return switch (f) { + .agent_model => "model", + .agent_transcript => "transcript", + else => @tagName(f), + }; + } +}; + +/// A resolved `/active` node. +pub const Act = struct { + file: AFile, + /// Only for `harness_dir`. + kind: active.Kind = .claude, + slot: u8 = 0, + gen: u8 = 0, + agent: u8 = 0, +}; + +pub fn activeNode(a: Act) u64 { + const serial: u48 = switch (a.file) { + .harness_dir => @intFromEnum(a.kind), + else => @as(u48, a.slot) | (@as(u48, a.gen) << 8) | (@as(u48, a.agent) << 16), + }; + return @bitCast(Node{ + .idx = @intFromEnum(a.file), + .kind = @intFromEnum(Kind.active), + .serial = serial, + }); +} + +pub const Node = packed struct(u64) { + /// A Top index (kind == .top) or a Root index (kind == .path). + idx: u8 = 0, + kind: u8 = 0, + serial: u48 = 0, +}; + +pub fn topNode(t: Top) u64 { + return @bitCast(Node{ .idx = @intFromEnum(t), .kind = @intFromEnum(Kind.top) }); +} + +pub const root: u64 = topNode(Top.root); + +/// One remembered mirrored file: its root, its relative path, the hash +/// that short-circuits dedupe lookups, and its reference count (one per +/// fid holding it, handed out by lookup, paid back by release). +const Entry = struct { + used: bool = false, + root: Root = .claude, + rel_buf: [rel_capacity]u8 = undefined, + rel_len: u16 = 0, + hash: u64 = 0, + refs: u32 = 0, + gen: u32 = 0, + + fn rel(e: *const Entry) []const u8 { + return e.rel_buf[0..e.rel_len]; + } +}; + +fn serialOf(slot: u12, gen: u32) u48 { + return (@as(u48, gen) << 12) | slot; +} + +fn nodeOf(e: *const Entry, slot: u12) u64 { + return @bitCast(Node{ + .idx = @intFromEnum(e.root), + .kind = @intFromEnum(Kind.path), + .serial = serialOf(slot, e.gen), + }); +} + +// ---- the exclusions (the security boundary) ----------------------------------- + +fn containsFold(name: []const u8, needle: []const u8) bool { + if (name.len < needle.len) return false; + var i: usize = 0; + while (i + needle.len <= name.len) : (i += 1) { + if (std.ascii.eqlIgnoreCase(name[i .. i + needle.len], needle)) return true; + } + return false; +} + +fn endsWithFold(name: []const u8, suffix: []const u8) bool { + return name.len >= suffix.len and std.ascii.eqlIgnoreCase(name[name.len - suffix.len ..], suffix); +} + +fn isOneOf(name: []const u8, comptime names: []const []const u8) bool { + inline for (names) |n| if (std.ascii.eqlIgnoreCase(name, n)) return true; + return false; +} + +/// Absolute rule: a name that may hold a credential, key, token or auth +/// material is never served, at any depth, by lookup or readdir. The +/// substrings are folded (case-insensitive) and deliberately broad — +/// "auth" also hides "author-notes", the price of never guessing wrong. +/// When unsure, exclude and document (docs/DESIGN.md). +pub fn excluded(name: []const u8) bool { + @setEvalBranchQuota(20000); + if (name.len == 0) return true; + for ([_][]const u8{ "credentials", "token", "auth", "secret" }) |needle| { + if (containsFold(name, needle)) return true; + } + if (endsWithFold(name, ".key")) return true; + if (endsWithFold(name, ".pem")) return true; + if (endsWithFold(name, ".env")) return true; + // Config and settings files of the other harnesses may embed API keys + // (hermes config.yaml, omp config.yml, dsh settings.yaml). Claude + // Code's settings.json is not in the tree at all: /claude serves only + // projects/, skills/ and history.jsonl. + if (isOneOf(name, &.{ "settings.json", "settings.yaml", "settings.local.json", "config.yml", "config.yaml", "config.toml", "models.yml", "models.yaml", ".ssh" })) return true; + for ([_][]const u8{ "id_rsa", "id_dsa", "id_ecdsa", "id_ed25519" }) |prefix| { + if (name.len >= prefix.len and std.ascii.eqlIgnoreCase(name[0..prefix.len], prefix)) return true; + } + return false; +} + +/// Every component of a relative path must pass `excluded`; used on the +/// comptime mount targets and the unit-test fixtures alike. +pub fn excludedPath(rel_path: []const u8) bool { + @setEvalBranchQuota(4000); + var it = std.mem.splitScalar(u8, rel_path, '/'); + while (it.next()) |comp| { + if (excluded(comp)) return true; + } + return false; +} + +// ---- the backend -------------------------------------------------------------- + +pub const Answer = struct { + reply: fs.Reply, + bytes: []const u8 = "", +}; + +fn fail(tag: u64, e: u16) Answer { + return .{ .reply = fs.Reply.fail(tag, e) }; +} + +/// A refusal that says why. 9P2000 has no numeric error — error(5) is +/// `Rerror tag[2] ename[s]`, a string — and acme names its refusals in words +/// ("permission denied", "file does not exist", fsys.c). A server that only +/// answers bare errnos throws away the one channel the protocol gives it for +/// explaining itself, which matters most here: a move has seven different +/// reasons to say no, and EPERM tells the user none of them. +/// +/// `errno` is still carried, because the reader may be a Linux mount rather +/// than a person: 9ns maps the *text* back to an errno by case-insensitive +/// substring (enameToErrno in 9ns/src/nine.zig, first match wins). So each +/// message below is worded to keep the phrase that carries its errno — +/// "not permitted" for EPERM, "invalid" for EINVAL, "exists" for EEXIST, +/// "no such" for ENOENT — and to avoid phrases that would silently retarget +/// it ("permission" and "denied" map to EACCES, "read-only" to EROFS, and +/// the bare substring "fid" to EBADF). +fn failWhy(tag: u64, e: u16, ename: []const u8) Answer { + return .{ .reply = .{ .tag = tag, .status = .err, .errno = e, .ename = ename } }; +} + +/// The daemon's state. One instance serves every connection; `handle` is +/// called with `mutex` held (the serve handler in main.zig holds it across +/// handle and reply, because `bytes` point into the shared buffers). +pub const Harness = struct { + pub const Req = fs.Req; + pub const Reply = fs.Reply; + pub const features: fs.Features = .{ .references = true }; + + io: Io, + pid: u32 = 0, + started_sec: i64 = 0, + /// The five pinned roots; a root without a base is unreachable. + base_buf: [5][base_capacity]u8 = @splat(@splat(0)), + base_len: [5]u16 = @splat(0), + base_set: [5]bool = @splat(false), + table: [path_capacity]Entry = @splat(.{}), + mutex: Io.Mutex = .init, + // Shared staging (guarded by `mutex`). + data_buf: [msize]u8 = undefined, + stage_buf: [msize]u8 = undefined, + list_names: [list_name_bytes]u8 = undefined, + list_dirs: [list_capacity]bool = undefined, + list_offs: [list_capacity]u32 = undefined, + + // The derived view (`/active`). An empty `proc` base turns it off, + // which is the default: a test must opt in to reading a process + // tree, and never the live one. + live: active.Table = .{}, + proc_buf: [base_capacity]u8 = @splat(0), + proc_len: u16 = 0, + home_buf: [base_capacity]u8 = @splat(0), + home_len: u16 = 0, + zmx_buf: [base_capacity]u8 = @splat(0), + zmx_len: u16 = 0, + runtime_buf: [base_capacity]u8 = @splat(0), + runtime_len: u16 = 0, + envp: ?cloud9.post.Env = null, + self_pid: u32 = 0, + btime: i64 = 0, + allow_move: bool = false, + act_text: [1024]u8 = undefined, + act_name: [64]u8 = undefined, + act_path: [active.path_capacity]u8 = undefined, + + pub const InitOptions = struct { + io: Io, + pid: u32, + /// Base path of each root, or "" when the root is not pinned. + bases: [5][]const u8, + /// The proc filesystem `/active` reads. Empty — the default — + /// leaves the derived view out of the tree entirely. + proc: []const u8 = "", + /// $HOME, for the harnesses' store-slug spellings. + home: []const u8 = "", + /// The zmx binary a move execs, resolved to an absolute path by + /// the caller. Never client bytes. + zmx: []const u8 = "zmx", + /// $XDG_RUNTIME_DIR, where a move looks for the zmx session it + /// is about to create. + runtime: []const u8 = "", + /// The daemon's own environment block, handed to the harness a + /// move re-execs. Null leaves a moved agent with an empty one. + envp: ?cloud9.post.Env = null, + /// Whether writing `/active///zmx` may move a session. + allow_move: bool = false, + }; + + pub fn init(h: *Harness, o: InitOptions) void { + h.* = .{ .io = o.io, .pid = o.pid, .started_sec = Io.Timestamp.now(o.io, .real).toSeconds() }; + if (o.proc.len > 0 and o.proc.len <= base_capacity) { + @memcpy(h.proc_buf[0..o.proc.len], o.proc); + h.proc_len = @intCast(o.proc.len); + h.self_pid = o.pid; + h.btime = active.bootTime(o.io, o.proc); + } + if (o.home.len <= base_capacity) { + @memcpy(h.home_buf[0..o.home.len], o.home); + h.home_len = @intCast(o.home.len); + } + if (o.zmx.len > 0 and o.zmx.len <= base_capacity) { + @memcpy(h.zmx_buf[0..o.zmx.len], o.zmx); + h.zmx_len = @intCast(o.zmx.len); + } + if (o.runtime.len <= base_capacity) { + @memcpy(h.runtime_buf[0..o.runtime.len], o.runtime); + h.runtime_len = @intCast(o.runtime.len); + } + h.allow_move = o.allow_move; + h.envp = o.envp; + inline for (0..5) |i| { + const path = o.bases[i]; + h.base_set[i] = path.len > 0 and path.len <= base_capacity; + if (h.base_set[i]) { + @memcpy(h.base_buf[i][0..path.len], path); + h.base_len[i] = @intCast(path.len); + } + } + } + + pub fn base(h: *const Harness, r: Root) ?[]const u8 { + if (!h.base_set[@intFromEnum(r)]) return null; + return h.base_buf[@intFromEnum(r)][0..h.base_len[@intFromEnum(r)]]; + } + + // -- path table -- + + fn hashOf(r: Root, rel_path: []const u8) u64 { + var hh = std.hash.Wyhash.init(0x6861_726e_6573_7300); // "harness" + hh.update(&.{@intFromEnum(r)}); + hh.update(rel_path); + return hh.final(); + } + + /// Finds or creates the entry for (root, rel). `ref` says whether the + /// caller hands out a reference (a lookup) or only names the file (a + /// readdir record). Two walks of the same file find the same entry, + /// so their node ids are equal; a freed entry's generation changes. + fn entryFor(h: *Harness, r: Root, rel_path: []const u8, ref: bool) ?*Entry { + if (rel_path.len == 0 or rel_path.len > rel_capacity) return null; + const hash = hashOf(r, rel_path); + var free: ?*Entry = null; + for (&h.table) |*e| { + if (!e.used) { + if (free == null) free = e; + continue; + } + if (e.hash == hash and e.root == r and e.rel_len == rel_path.len and + std.mem.eql(u8, e.rel(), rel_path)) + { + if (ref) e.refs += 1; + return e; + } + } + const e = free orelse return null; + e.used = true; + e.root = r; + e.rel_len = @intCast(rel_path.len); + @memcpy(e.rel_buf[0..rel_path.len], rel_path); + e.hash = hash; + e.refs = @intFromBool(ref); + e.gen +%= 1; + return e; + } + + fn slotOf(h: *Harness, e: *Entry) u12 { + const off = (@intFromPtr(e) - @intFromPtr(&h.table)) / @sizeOf(Entry); + return @intCast(off); + } + + /// The target a node names; a freed or forged serial answers null. + fn resolve(h: *Harness, node: u64) ?Target { + const n: Node = @bitCast(node); + switch (std.enums.fromInt(Kind, n.kind) orelse return null) { + .top => return .{ .top = std.enums.fromInt(Top, n.idx) orelse return null }, + .path => { + const r = std.enums.fromInt(Root, n.idx) orelse return null; + const slot: u12 = @truncate(n.serial); + const e = &h.table[slot]; + if (!e.used or e.root != r) return null; + if (serialOf(slot, e.gen) != n.serial) return null; + return .{ .path = e }; + }, + .active => { + const f = std.enums.fromInt(AFile, n.idx) orelse return null; + if (f == .harness_dir) { + const k = std.enums.fromInt(active.Kind, @as(u8, @truncate(n.serial))) orelse return null; + return .{ .act = .{ .file = f, .kind = k } }; + } + const slot: u8 = @truncate(n.serial); + const gen: u8 = @truncate(n.serial >> 8); + const agent: u8 = @truncate(n.serial >> 16); + const l = h.live.at(slot, gen) orelse return null; + return .{ .act = .{ .file = f, .kind = l.kind, .slot = slot, .gen = gen, .agent = agent } }; + }, + } + } + + fn entryNode(h: *Harness, e: *Entry) u64 { + return nodeOf(e, h.slotOf(e)); + } +}; + +pub const Target = union(enum) { + top: Top, + path: *Entry, + act: Act, +}; + +// ---- attributes --------------------------------------------------------------- + +/// Opens `/` without following a symlink at any step +/// below the base: every component is opened with `O_NOFOLLOW`, so a +/// name swapped for a symlink between the walk and the read fails +/// instead of reaching out of the root. Composing the whole path and +/// opening it in one call cannot do that — only the last component's +/// symlink is refused then, and every directory above it is followed +/// wherever it points. The base itself is configuration (`--root` or +/// $HOME), not client bytes, so it resolves normally. The caller owns +/// the descriptor. +fn openIn(h: *Harness, r: Root, rel_path: []const u8, final: linux.O) ?i32 { + const b = h.base(r) orelse return null; + if (rel_path.len > rel_capacity or b.len > base_capacity) return null; + const walk: linux.O = .{ .PATH = true, .DIRECTORY = true, .CLOEXEC = true, .NOFOLLOW = true }; + + var base_z: [base_capacity + 1]u8 = @splat(0); + @memcpy(base_z[0..b.len], b); + var top_flags = if (rel_path.len == 0) final else walk; + top_flags.NOFOLLOW = false; // the pinned root may legitimately be a link + top_flags.CLOEXEC = true; + var fd = fdOf(linux.open(@ptrCast(&base_z), top_flags, 0)) orelse return null; + if (rel_path.len == 0) return fd; + + var rest = rel_path; + while (true) { + const slash = std.mem.indexOfScalar(u8, rest, '/'); + const comp = if (slash) |i| rest[0..i] else rest; + const last = slash == null; + // Nothing below composes these; a component that could re-enter + // the walk is refused rather than opened. + if (comp.len == 0 or comp.len > 255 or + std.mem.eql(u8, comp, ".") or std.mem.eql(u8, comp, "..")) + { + _ = linux.close(fd); + return null; + } + var name_z: [256]u8 = @splat(0); + @memcpy(name_z[0..comp.len], comp); + var flags = if (last) final else walk; + flags.NOFOLLOW = true; + flags.CLOEXEC = true; + const next = fdOf(linux.openat(fd, @ptrCast(&name_z), flags, 0)); + _ = linux.close(fd); + fd = next orelse return null; + if (last) return fd; + rest = rest[comp.len + 1 ..]; + } +} + +fn fdOf(rc: usize) ?i32 { + if (linux.errno(rc) != .SUCCESS) return null; + return @intCast(rc); +} + +/// Stat of `/`, resolved the no-follow way. A symlink +/// is never served: it is the one escape hatch around the pinned roots. +/// `O_PATH` opens the name itself, so a fifo or device never blocks and +/// never has its open side effects run. +/// A stat plus the file's identity on disk. +const Stat = struct { + st: Io.Dir.Stat, + /// The qid path to report, or 0 when the kernel would not say. + /// + /// intro(5) is strict about what a qid path means: "The path is an + /// integer unique among all files in the hierarchy. If a file is deleted + /// and recreated with the same name in the same directory, the old and + /// new path components of the qids should be different" — and two files + /// are the same file if and only if their qids match. So the path has to + /// be a property of the *file*, stable for as long as it exists. + /// + /// The backend's own node id cannot supply that: it is a slot in the path + /// table, and a slot freed by its last release comes back with a new + /// generation (the guard that stops a forged node id resolving). A client + /// that walks, stats and clunks — `9p stat` does exactly this — saw a + /// different qid path for the same unchanged file every single time, and + /// by the protocol's own rule that means a different file. + /// + /// So this is u9fs's answer instead, from the same fstat the attr already + /// needs: + /// + /// memmove(p, &st->st_ino, ...); *(dev_t*)q ^= st->st_dev; -- u9fs.c + /// + /// The device is mixed in because an inode number is only unique within + /// its filesystem, and the five pinned roots need not share one. The + /// engine keeps resolving requests by node id, which still recycles; only + /// the qid the client sees is stabilised (fs.Attr.path exists for exactly + /// this split, and the engine uses it for nothing else). + qid_path: u64, +}; + +fn statIn(h: *Harness, r: Root, rel_path: []const u8) ?Stat { + const fd = openIn(h, r, rel_path, .{ .PATH = true, .CLOEXEC = true }) orelse return null; + const f: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + defer f.close(h.io); + const st = f.stat(h.io) catch return null; + if (st.kind == .sym_link) return null; // never served: an escape hatch + return .{ .st = st, .qid_path = identityOf(fd) }; +} + +/// (dev, ino) of an open descriptor, folded into one integer. 0 when statx +/// declines to answer, which the caller reads as "fall back to the node id". +fn identityOf(fd: i32) u64 { + var sx: linux.Statx = undefined; + const want: linux.STATX = .{ .INO = true }; + const rc = linux.statx(fd, "", linux.AT.EMPTY_PATH, want, &sx); + if (linux.errno(rc) != .SUCCESS or !sx.mask.INO) return 0; + const dev = (@as(u64, sx.dev_major) << 32) | sx.dev_minor; + // The inode carries the entropy, so it stays unshifted and the device is + // folded over its top bits: two roots on one filesystem keep distinct + // inodes, and the same inode on two filesystems stops colliding. + return sx.ino ^ (dev *% 0x9E37_79B9_7F4A_7C15); +} + +/// The qid version of a file on disk, the way u9fs derives it: +/// +/// qid.vers = st->st_mtime ^ (st->st_size << 8); -- u9fs.c +/// +/// intro(5) makes the qid the server's identity for a file — "two files on +/// the same server hierarchy are the same if and only if their qids are the +/// same" — and the version the part that "is incremented every time the file +/// is modified". A server that leaves it 0 is telling every client that +/// nothing it serves ever changes, which is exactly wrong for this tree: its +/// whole point is transcripts that grow while you read them. The size shift +/// is what makes an append within the same second still change the version, +/// and a rewrite that keeps the size change it through the mtime. +fn qidVers(mtime_sec: i64, size: u64) u32 { + return @truncate(@as(u64, @bitCast(mtime_sec)) ^ (size << 8)); +} + +/// The qid version of text this server derives rather than reads: there is no +/// mtime behind it, so the content itself is the only honest version. +fn textVers(text: []const u8) u32 { + return @truncate(std.hash.Wyhash.hash(0, text)); +} + +fn statAttr(h: *Harness, e: *Entry) ?fs.Attr { + const got = statIn(h, e.root, e.rel()) orelse return null; + const st = got.st; + const mtime = st.mtime.toSeconds(); + const size: u64 = if (st.kind == .directory) 0 else st.size; + return .{ + .name = basename(e.rel()), + .node = h.entryNode(e), + .dir = st.kind == .directory, + // stat(5): "Directories and most files representing devices have a + // conventional length of 0." + .size = size, + .mode = if (st.kind == .directory) 0o555 else 0o444, + .mtime = @truncate(@as(u64, @bitCast(mtime))), + .version = qidVers(mtime, size), + .path = if (got.qid_path != 0) got.qid_path else null, + }; +} + +fn basename(rel_path: []const u8) []const u8 { + if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| return rel_path[i + 1 ..]; + return rel_path; +} + +/// Served as `/README`: what the tree is and how to read it. Static, so it +/// is the one "fact" that needs no buffer — `factText` returns it directly. +const readme_text = + \\9agents - every coding agent's state on this machine, as files. + \\Read-only: writes answer EPERM. Nothing is cached, so a transcript + \\grows while you cat it and a new session appears as soon as its + \\file lands. + \\ + \\ README this file + \\ pid uptime this daemon's pid; seconds since it started + \\ active/ what is running right now + \\ claude/ projects/ history skills/ + \\ codex/ sessions/ session-index history + \\ omp/ hermes/ dsh/ each harness's own directory, raw + \\ skills/ claude/ codex/ omp/, in one place + \\ + \\active/// holds pid ppid started cwd name title + \\session via status transcript zmx. + \\ session its session id, empty when it did not resolve + \\ via how that id was found: registry, fd, dir or none + \\ status only harnesses that publish one have it + \\ transcript the live .jsonl; its mtime is the last activity + \\ zmx the zmx session it runs in, empty outside zmx + \\ + \\ ls active/*/* what is running + \\ cat active/claude/345104/session what it is + \\ tail -f active/claude/345104/transcript + \\ + \\Credentials are excluded by name and never listed, at any depth; + \\symlinks are never served. Writing a name into an agent's zmx file + \\moves it into that zmx session, and only when the daemon was + \\started with --allow-move; otherwise that write answers EPERM too. + \\ +; + +/// A fact's text, rendered on demand. `buf` should be a small stack buffer. +fn factText(h: *Harness, fact: Top, buf: []u8) []const u8 { + if (fact == .readme) return readme_text; // static: outlives any buffer + var w = Io.Writer.fixed(buf); + switch (fact) { + .pid => w.print("{d}\n", .{h.pid}) catch {}, + .uptime => { + const now = Io.Timestamp.now(h.io, .real).toSeconds(); + const up: u64 = if (now > h.started_sec) @intCast(now - h.started_sec) else 0; + w.print("{d}\n", .{up}) catch {}; + }, + else => {}, + } + return w.buffered(); +} + +var fact_buf: [64]u8 = undefined; + +fn attrFor(h: *Harness, t: Target) ?fs.Attr { + switch (t) { + .top => |top| { + switch (top) { + .root => return .{ .name = "/", .node = root, .dir = true, .mode = 0o555 }, + .active => return .{ .name = "active", .node = topNode(.active), .dir = true, .mode = 0o555 }, + .pid, .uptime, .readme => return .{ + .name = top.fileName(), + .node = topNode(top), + .size = factText(h, top, &fact_buf).len, + .mode = 0o444, + }, + .claude, .codex, .skills => return .{ + .name = top.fileName(), + .node = topNode(top), + .dir = true, + .mode = 0o555, + }, + .omp, .hermes, .dsh => { + // The harness dir IS its mirror: stat the target. + const m = top.mirror().?; + const got = statIn(h, m.root, m.rel) orelse return null; + if (got.st.kind != .directory) return null; + return .{ + .name = top.fileName(), + .node = topNode(top), + .dir = true, + .mode = 0o555, + .path = if (got.qid_path != 0) got.qid_path else null, + }; + }, + } + }, + .path => |e| return statAttr(h, e), + .act => |a| return activeAttr(h, a), + } +} + +/// Attr of the mount target as a table entry (fresh from disk, or null +/// when the target is missing, a symlink, or excluded by name). +fn attrOfMount(h: *Harness, r: Root, rel_path: []const u8) ?fs.Attr { + const e = h.entryFor(r, rel_path, false) orelse return null; + defer if (e.refs == 0) { + e.used = false; + e.gen +%= 1; + }; + return statAttr(h, e); +} + +// ---- /active: the derived view ------------------------------------------------- + +fn sourcesOf(h: *Harness) active.Sources { + var roots: [5][]const u8 = @splat(""); + for (0..5) |i| { + if (h.base_set[i]) roots[i] = h.base_buf[i][0..h.base_len[i]]; + } + return .{ + .io = h.io, + .proc = h.proc_buf[0..h.proc_len], + .home = h.home_buf[0..h.home_len], + .roots = roots, + .self_pid = h.self_pid, + .btime = h.btime, + }; +} + +/// Rescans the process tree. Done on a listing of `/active` and on a +/// lookup that misses, which is what makes an agent visible the moment +/// it starts and gone the moment it exits. +fn rescan(h: *Harness) void { + if (h.proc_len == 0) return; + h.live.scan(sourcesOf(h)); +} + +fn hasLive(h: *Harness, k: active.Kind) bool { + for (&h.live.slots) |*sl| { + if (sl.used and sl.live.kind == k) return true; + } + return false; +} + +/// `\n` staged where an answer can point at it. +fn line(h: *Harness, text: []const u8) ?[]const u8 { + return std.fmt.bufPrint(&h.act_text, "{s}\n", .{text}) catch null; +} + +/// The value of a field file, or null when this agent has none — an +/// absent field is not listed and does not resolve, so the tree never +/// answers a blank where it does not know. +fn activeText(h: *Harness, a: Act) ?[]const u8 { + const l = h.live.at(a.slot, a.gen) orelse return null; + const src = sourcesOf(h); + var scratch: [active.text_capacity]u8 = undefined; + return switch (a.file) { + .pid => std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.pid}) catch null, + .ppid => if (l.ppid == 0) null else std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.ppid}) catch null, + .started => if (l.started == 0) null else std.fmt.bufPrint(&h.act_text, "{d}\n", .{l.started}) catch null, + .cwd => if (l.cwd.len == 0) null else line(h, l.cwd.slice()), + .session => if (l.session.len == 0) null else line(h, l.session.slice()), + .via => line(h, l.via.text()), + .name => line(h, active.registryField(src, l, "name", &scratch) orelse return null), + .status => line(h, active.registryField(src, l, "status", &scratch) orelse return null), + .title => line(h, active.headField(h.io, l.kind, l.transcript.slice(), .title, &scratch) orelse return null), + .model => line(h, active.headField(h.io, l.kind, l.transcript.slice(), .model, &scratch) orelse return null), + .agent_model => blk: { + const path = activePath(h, a) orelse break :blk null; + break :blk line(h, active.headField(h.io, l.kind, path, .model, &scratch) orelse return null); + }, + // `zmx` is always there, so it can always be written to; empty + // means the agent runs outside zmx. + .zmx => if (active.zmxOf(src, l.pid, &scratch)) |v| line(h, v) else h.act_text[0..0], + else => null, + }; +} + +/// The absolute path behind a transcript-shaped file. +fn activePath(h: *Harness, a: Act) ?[]const u8 { + const l = h.live.at(a.slot, a.gen) orelse return null; + switch (a.file) { + .transcript => return if (l.transcript.len == 0) null else l.transcript.slice(), + .agent_transcript, .agent_model => { + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + return active.agentPath(l, &ag, a.agent, &h.act_path); + }, + else => return null, + } +} + +/// Opens an absolute path without following a symlink at the last +/// component. These paths are derived from a pinned root or from the +/// fd the harness itself holds, never from client bytes, but a name +/// swapped underneath must still fail rather than redirect. +fn openAbsNoFollow(path: []const u8) ?i32 { + if (path.len == 0 or path.len >= active.path_capacity) return null; + var z: [active.path_capacity]u8 = @splat(0); + @memcpy(z[0..path.len], path); + const rc = linux.open(@ptrCast(&z), .{ + .ACCMODE = .RDONLY, + .NOFOLLOW = true, + .CLOEXEC = true, + .NONBLOCK = true, + }, 0); + if (linux.errno(rc) != .SUCCESS) return null; + return @intCast(rc); +} + +fn activeAttr(h: *Harness, a: Act) ?fs.Attr { + switch (a.file) { + .harness_dir => { + if (!hasLive(h, a.kind)) return null; + return .{ .name = a.kind.text(), .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .entry => { + const l = h.live.at(a.slot, a.gen) orelse return null; + const nm = std.fmt.bufPrint(&h.act_name, "{d}", .{l.pid}) catch return null; + return .{ .name = nm, .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .agents => { + const l = h.live.at(a.slot, a.gen) orelse return null; + if (l.agent_dir.len == 0) return null; + return .{ .name = "agents", .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .agent => { + const l = h.live.at(a.slot, a.gen) orelse return null; + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + if (a.agent >= ag.count) return null; + const nm = ag.name(a.agent); + @memcpy(h.act_name[0..nm.len], nm); + return .{ .name = h.act_name[0..nm.len], .node = activeNode(a), .dir = true, .mode = 0o555 }; + }, + .transcript, .agent_transcript => { + const path = activePath(h, a) orelse return null; + const st = Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = false }) catch return null; + if (st.kind != .file) return null; + const mtime = st.mtime.toSeconds(); + return .{ + .name = a.file.fileName(), + .node = activeNode(a), + .size = st.size, + .mode = 0o444, + .mtime = @truncate(@as(u64, @bitCast(mtime))), + .version = qidVers(mtime, st.size), + }; + }, + else => { + const text = activeText(h, a) orelse return null; + // Only `zmx` is ever writable, and only when a move is allowed. + const mode: u16 = if (a.file == .zmx and h.allow_move) 0o644 else 0o444; + return .{ + .name = a.file.fileName(), + .node = activeNode(a), + .size = text.len, + .mode = mode, + // Derived, not read: `status` flipping busy->idle keeps its + // size, so only the text can carry the change. + .version = textVers(text), + }; + }, + } +} + +/// The slot holding `pid` for this harness, if any. +fn slotOfPid(h: *Harness, k: active.Kind, pid: u32) ?Act { + for (&h.live.slots, 0..) |*sl, i| { + if (!sl.used or sl.live.kind != k or sl.live.pid != pid) continue; + return .{ .file = .entry, .kind = k, .slot = @intCast(i), .gen = sl.gen }; + } + return null; +} + +fn activeLookup(h: *Harness, req: fs.Req, a: Act, name: []const u8) Answer { + switch (a.file) { + .harness_dir => { + const pid = std.fmt.parseInt(u32, name, 10) catch return fail(req.tag, E.NOENT); + var found = slotOfPid(h, a.kind, pid); + if (found == null) { + rescan(h); + found = slotOfPid(h, a.kind, pid); + } + const act = found orelse return fail(req.tag, E.NOENT); + return activeReply(h, req, act, name); + }, + .entry => { + if (std.mem.eql(u8, name, "agents")) { + return activeReply(h, req, .{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen }, name); + } + for (AFile.entry_files) |f| { + if (!std.mem.eql(u8, f.fileName(), name)) continue; + return activeReply(h, req, .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen }, name); + } + return fail(req.tag, E.NOENT); + }, + .agents => { + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + const i = ag.indexOf(name) orelse return fail(req.tag, E.NOENT); + return activeReply(h, req, .{ + .file = .agent, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = @intCast(i), + }, name); + }, + .agent => { + for (AFile.agent_files) |f| { + if (!std.mem.eql(u8, f.fileName(), name)) continue; + return activeReply(h, req, .{ + .file = f, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = a.agent, + }, name); + } + return fail(req.tag, E.NOENT); + }, + else => return fail(req.tag, E.NOTDIR), + } +} + +fn activeReply(h: *Harness, req: fs.Req, a: Act, name: []const u8) Answer { + const attr = activeAttr(h, a) orelse return fail(req.tag, E.NOENT); + var with_name = attr; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; +} + +fn activeReaddir(h: *Harness, req: fs.Req, a: Act) Answer { + var st: Staging = .{ .buf = &h.stage_buf, .skip = req.off }; + switch (a.file) { + .harness_dir => { + for (&h.live.slots, 0..) |*sl, i| { + if (!sl.used or sl.live.kind != a.kind) continue; + var num: [24]u8 = undefined; + const nm = std.fmt.bufPrint(&num, "{d}", .{sl.live.pid}) catch continue; + st.add(activeNode(.{ + .file = .entry, + .kind = a.kind, + .slot = @intCast(i), + .gen = sl.gen, + }), true, nm); + } + }, + .entry => { + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + for (AFile.entry_files) |f| { + const child: Act = .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen }; + if (activeAttr(h, child) == null) continue; // no value: not listed + st.add(activeNode(child), false, f.fileName()); + } + if (l.agent_dir.len > 0) { + st.add(activeNode(.{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen }), true, "agents"); + } + }, + .agents => { + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + var ag: active.Agents = .{}; + active.agentsOf(h.io, l, &ag); + for (0..ag.count) |i| { + st.add(activeNode(.{ + .file = .agent, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = @intCast(i), + }), true, ag.name(i)); + } + }, + .agent => { + for (AFile.agent_files) |f| { + const child: Act = .{ .file = f, .kind = a.kind, .slot = a.slot, .gen = a.gen, .agent = a.agent }; + if (activeAttr(h, child) == null) continue; + st.add(activeNode(child), false, f.fileName()); + } + }, + else => return fail(req.tag, E.NOTDIR), + } + return .{ .reply = .{ .tag = req.tag }, .bytes = h.stage_buf[0..st.len] }; +} + +fn activeRead(h: *Harness, req: fs.Req, a: Act) Answer { + switch (a.file) { + .harness_dir, .entry, .agents, .agent => return fail(req.tag, E.ISDIR), + .transcript, .agent_transcript => { + const path = activePath(h, a) orelse return fail(req.tag, E.NOENT); + const fd = openAbsNoFollow(path) orelse return fail(req.tag, E.NOENT); + const file: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + defer file.close(h.io); + const st = file.stat(h.io) catch return fail(req.tag, E.IO); + if (st.kind == .directory) return fail(req.tag, E.ISDIR); + if (st.kind != .file) return fail(req.tag, E.PERM); + _ = linux.fcntl(fd, linux.F.SETFL, 0); + const want = @min(req.size, h.data_buf.len); + const n = file.readPositionalAll(h.io, h.data_buf[0..want], req.off) catch + return fail(req.tag, E.IO); + return .{ .reply = .{ .tag = req.tag }, .bytes = h.data_buf[0..n] }; + }, + else => { + const text = activeText(h, a) orelse return fail(req.tag, E.NOENT); + return window(req, text); + }, + } +} + +// ---- the move: writing a zmx session name into an agent's `zmx` ---------------- + +/// How long a move waits, in 100ms ticks: for the agent to take the +/// hangup, then to die outright, then for the new zmx session to post. +const term_ticks: usize = 50; +const kill_ticks: usize = 30; +const post_ticks: usize = 100; + +fn napOneTick(h: *Harness) void { + h.io.sleep(.fromMilliseconds(100), .awake) catch {}; +} + +/// Is this name already a live zmx session? zmx posts each session into +/// the registry, so the registry is the answer — no process scanning. +fn zmxPosted(h: *Harness, name: []const u8) bool { + if (h.runtime_len == 0) return false; + var buf: [active.path_capacity]u8 = undefined; + const path = std.fmt.bufPrint(&buf, "{s}/9p/zmx/{s}", .{ h.runtime_buf[0..h.runtime_len], name }) catch return true; + return Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = true }) != error.FileNotFound; +} + +fn procGone(h: *Harness, pid: u32) bool { + var buf: [active.path_capacity]u8 = undefined; + const path = std.fmt.bufPrint(&buf, "{s}/{d}", .{ h.proc_buf[0..h.proc_len], pid }) catch return false; + _ = Io.Dir.statFile(.cwd(), h.io, path, .{ .follow_symlinks = true }) catch return true; + return false; +} + +/// SIGTERM, then SIGKILL, then give up. The agent's session was +/// resolved before this ran, so whatever happens it has somewhere to +/// come back to. +fn killAndWait(h: *Harness, pid: u32) bool { + _ = linux.kill(@intCast(pid), .TERM); + for (0..term_ticks) |_| { + if (procGone(h, pid)) return true; + napOneTick(h); + } + _ = linux.kill(@intCast(pid), .KILL); + for (0..kill_ticks) |_| { + if (procGone(h, pid)) return true; + napOneTick(h); + } + return false; +} + +/// `zmx run -d `, in the +/// agent's own directory. Every word but `` is fixed by the +/// harness; `` was checked against zmx's label charset before +/// anything was killed. No byte a client wrote reaches `exec` as a +/// command. +fn spawnZmx(h: *Harness, l: *const active.Live, name: []const u8, resume_value: []const u8) bool { + if (h.zmx_len == 0) return false; + const harness_argv0: []const u8 = @tagName(l.kind); + const resume_flag: []const u8 = switch (l.kind) { + .codex => "resume", + else => "--resume", + }; + + // One buffer holds every NUL-terminated word; `argv` points into it. + var words: [8 * active.path_capacity]u8 = undefined; + var used: usize = 0; + var argv: [9]?[*:0]const u8 = @splat(null); + var n: usize = 0; + const parts = [_][]const u8{ + h.zmx_buf[0..h.zmx_len], "run", name, "-d", + harness_argv0, resume_flag, resume_value, + }; + for (parts) |w| { + if (used + w.len + 1 > words.len or n + 1 >= argv.len) return false; + @memcpy(words[used..][0..w.len], w); + words[used + w.len] = 0; + argv[n] = @ptrCast(&words[used]); + n += 1; + used += w.len + 1; + } + + var cwd_z: [active.cwd_capacity + 1]u8 = @splat(0); + if (l.cwd.len >= cwd_z.len) return false; + @memcpy(cwd_z[0..l.cwd.len], l.cwd.slice()); + + const rc = linux.fork(); + if (linux.errno(rc) != .SUCCESS) return false; + if (rc == 0) { + // The child: only async-signal-safe calls from here. + _ = linux.chdir(@ptrCast(&cwd_z)); + const empty = [_:null]?[*:0]const u8{}; + const envp: cloud9.post.Env = h.envp orelse ∅ + _ = linux.execve(argv[0].?, @ptrCast(&argv), envp); + linux.exit(127); + } + // Reap the forked `zmx`, which returns as soon as the session is up. + const child: i32 = @intCast(rc); + var status: u32 = 0; + for (0..post_ticks) |_| { + const w = linux.waitpid(child, &status, 1); // WNOHANG + if (w == @as(usize, @intCast(child))) break; + napOneTick(h); + } + return true; +} + +/// A move: kill the agent and bring it back inside a zmx session of the +/// name written. Refusals come before anything is destroyed. +fn moveToZmx(h: *Harness, req: fs.Req, a: Act) Answer { + if (!h.allow_move) return failWhy(req.tag, E.PERM, "moves are not permitted: start 9agents with --allow-move"); + const name = std.mem.trim(u8, req.data, " \t\r\n"); + if (!active.legalZmxName(name)) return failWhy(req.tag, E.INVAL, "invalid zmx name: letters, digits, dot, dash and underscore only"); + + const l = h.live.at(a.slot, a.gen) orelse return fail(req.tag, E.NOENT); + // Nothing is killed that has nowhere to come back to. + if (l.via == .none or l.session.len == 0) return failWhy(req.tag, E.PERM, "no session resolved: not permitted, there would be nothing to resume"); + if (l.cwd.len == 0) return failWhy(req.tag, E.PERM, "no working directory known for this agent: not permitted"); + const resume_value: []const u8 = switch (l.kind) { + // omp resumes by the transcript it wrote, the others by id. + .omp => l.transcript.slice(), + .claude, .codex, .hermes => l.session.slice(), + // dsh has no resume form worth guessing at. + .dsh => return failWhy(req.tag, E.PERM, "dsh has no resume form: moving it is not permitted"), + }; + if (resume_value.len == 0) return failWhy(req.tag, E.PERM, "nothing to resume from: not permitted"); + + // Already there: setting a value it already has changes nothing. + var have: [active.text_capacity]u8 = undefined; + if (active.zmxOf(sourcesOf(h), l.pid, &have)) |current| { + if (std.mem.eql(u8, current, name)) { + return .{ .reply = .{ .tag = req.tag, .written = @intCast(req.data.len) } }; + } + } + if (zmxPosted(h, name)) return failWhy(req.tag, E.EXIST, "a live zmx session of that name already exists"); + // The slot could have gone stale between the scan and this write. + if (!active.stillAlive(sourcesOf(h), l)) return failWhy(req.tag, E.NOENT, "the agent exited before the move began: no such process"); + + // Everything below this line destroys something. + var snapshot = l.*; + if (!killAndWait(h, snapshot.pid)) return failWhy(req.tag, E.IO, "the agent did not exit when signalled; nothing was started"); + if (!spawnZmx(h, &snapshot, name, resume_value)) return failWhy(req.tag, E.IO, "could not start zmx; the agent has already been stopped"); + for (0..post_ticks) |_| { + if (zmxPosted(h, name)) { + rescan(h); + return .{ .reply = .{ .tag = req.tag, .written = @intCast(req.data.len) } }; + } + napOneTick(h); + } + rescan(h); + return failWhy(req.tag, E.IO, "zmx did not post the session in time; the agent may still return"); +} + +// ---- dispatch ------------------------------------------------------------------ + +/// Answers one engine request. The caller holds `h.mutex` and replies +/// before letting go of it (answer bytes point into `h`). +pub fn handle(h: *Harness, req: fs.Req) Answer { + const t = h.resolve(req.node) orelse return fail(req.tag, E.NOENT); + return switch (req.op) { + .lookup => lookup(h, req, t), + .getattr => attrReply(h, req.tag, t), + .setattr => setattrReq(h, req, t), + .open => open(h, req, t), + .release => release(h, req, t), + .readdir => readdir(h, req, t), + .read => read(h, req, t), + .write => writeReq(h, req, t), + }; +} + +/// Truncation is how a shell's `>` opens a file before writing it. The +/// `zmx` file has no length of its own — a write replaces the value — +/// so on the one writable file a zero-length truncate is a no-op +/// rather than a refusal. Everything else still answers EPERM. +fn setattrReq(h: *Harness, req: fs.Req, t: Target) Answer { + switch (t) { + .act => |a| if (a.file == .zmx and h.allow_move and req.truncate) { + return .{ .reply = .{ .tag = req.tag } }; + }, + else => {}, + } + return failWhy(req.tag, E.PERM, "9agents serves a view, not a store: writing is not permitted"); +} + +/// The tree answers EPERM to every write but one: a zmx session name +/// into a live agent's `zmx`, which moves it there. +fn writeReq(h: *Harness, req: fs.Req, t: Target) Answer { + switch (t) { + .act => |a| if (a.file == .zmx) return moveToZmx(h, req, a), + else => {}, + } + return failWhy(req.tag, E.PERM, "9agents serves a view, not a store: writing is not permitted"); +} + +fn attrReply(h: *Harness, tag: u64, t: Target) Answer { + const a = attrFor(h, t) orelse return fail(tag, E.NOENT); + return .{ .reply = .{ .tag = tag, .attr = a } }; +} + +/// A lookup's attr names the file as the client asked for it (the engine +/// keeps that name for the fid); refs for path targets were already taken +/// by `entryFor`. +fn lookupAttr(h: *Harness, req: fs.Req, t: Target, name: []const u8) Answer { + const a = attrFor(h, t) orelse return fail(req.tag, E.NOENT); + var with_name = a; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; +} + +fn lookup(h: *Harness, req: fs.Req, t: Target) Answer { + const name = req.data; + if (name.len == 0 or name.len > 255) return fail(req.tag, E.NOENT); + if (std.mem.indexOfAny(u8, name, "/\x00") != null) return fail(req.tag, E.NOENT); + + if (std.mem.eql(u8, name, ".")) { + // The engine asks for "." when cloning a fid; the reference is + // paid for path targets here like any other lookup result. + switch (t) { + .top, .act => return attrReply(h, req.tag, t), + .path => |e| { + e.refs += 1; + return lookupAttr(h, req, t, name); + }, + } + } + if (std.mem.eql(u8, name, "..")) return lookupParent(h, req, t); + + switch (t) { + .top => |top| switch (top) { + .root => { + const found: ?Top = blk: for (Top.listed) |e| { + if (std.mem.eql(u8, e.fileName(), name)) break :blk e; + } else break :blk null; + return attrReply(h, req.tag, .{ .top = found orelse return fail(req.tag, E.NOENT) }); + }, + .claude, .codex, .skills => { + const m = mountNamed(top, name) orelse return fail(req.tag, E.NOENT); + const e = h.entryFor(m.root, m.rel, true) orelse return fail(req.tag, E.NFILE); + const a = statAttr(h, e) orelse { + unref(e); + return fail(req.tag, E.NOENT); + }; + var with_name = a; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; + }, + .omp, .hermes, .dsh => { + const m = top.mirror().?; + return lookupBelow(h, req, m.root, m.rel, name); + }, + .active => { + const k = blk: for (std.enums.values(active.Kind)) |k| { + if (std.mem.eql(u8, k.text(), name)) break :blk k; + } else return fail(req.tag, E.NOENT); + if (!hasLive(h, k)) { + rescan(h); + if (!hasLive(h, k)) return fail(req.tag, E.NOENT); + } + return activeReply(h, req, .{ .file = .harness_dir, .kind = k }, name); + }, + .pid, .uptime, .readme => return fail(req.tag, E.NOTDIR), + }, + .path => |e| { + const rel_path = e.rel(); + if (attrFor(h, t)) |a| { + if (!a.dir) return fail(req.tag, E.NOTDIR); + } else return fail(req.tag, E.NOENT); + return lookupBelow(h, req, e.root, rel_path, name); + }, + .act => |a| return activeLookup(h, req, a, name), + } +} + +/// A lookup of `name` in the directory (root, dir_rel): the child's rel +/// path is composed only from the remembered pair and the (single-segment, +/// engine-vetted) name. +fn lookupBelow(h: *Harness, req: fs.Req, r: Root, dir_rel: []const u8, name: []const u8) Answer { + if (excluded(name)) return fail(req.tag, E.NOENT); + var scratch: [rel_capacity + 256]u8 = undefined; + const child_rel = joinRel(&scratch, dir_rel, name) orelse return fail(req.tag, E.NOENT); + const e = h.entryFor(r, child_rel, true) orelse return fail(req.tag, E.NFILE); + const a = statAttr(h, e) orelse { + unref(e); + return fail(req.tag, E.NOENT); + }; + var with_name = a; + with_name.name = name; + return .{ .reply = .{ .tag = req.tag, .attr = with_name } }; +} + +fn joinRel(scratch: []u8, dir_rel: []const u8, name: []const u8) ?[]const u8 { + if (dir_rel.len == 0) { + if (name.len > scratch.len) return null; + @memcpy(scratch[0..name.len], name); + return scratch[0..name.len]; + } + const total = dir_rel.len + 1 + name.len; + if (total > scratch.len) return null; + @memcpy(scratch[0..dir_rel.len], dir_rel); + scratch[dir_rel.len] = '/'; + @memcpy(scratch[dir_rel.len + 1 .. total], name); + return scratch[0..total]; +} + +fn lookupParent(h: *Harness, req: fs.Req, t: Target) Answer { + const parent: Target = switch (t) { + .top => |top| .{ .top = top.parent() }, + .path => |e| blk: { + const rel_path = e.rel(); + if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| { + const up = rel_path[0..i]; + const pe = h.entryFor(e.root, up, true) orelse return fail(req.tag, E.NFILE); + break :blk .{ .path = pe }; + } + break :blk .{ .top = mountOwner(e.root, rel_path) }; + }, + .act => |a| switch (a.file) { + .harness_dir => .{ .top = .active }, + .entry => .{ .act = .{ .file = .harness_dir, .kind = a.kind } }, + .agents => .{ .act = .{ .file = .entry, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, + .agent => .{ .act = .{ .file = .agents, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, + .agent_model, .agent_transcript => .{ .act = .{ + .file = .agent, + .kind = a.kind, + .slot = a.slot, + .gen = a.gen, + .agent = a.agent, + } }, + // Every other file hangs directly off its entry. + else => .{ .act = .{ .file = .entry, .kind = a.kind, .slot = a.slot, .gen = a.gen } }, + }, + }; + return attrReply(h, req.tag, parent); +} + +fn unref(e: *Entry) void { + if (e.refs > 0) e.refs -= 1; + if (e.refs == 0) { + e.used = false; + e.gen +%= 1; + } +} + +fn open(h: *Harness, req: fs.Req, t: Target) Answer { + _ = attrFor(h, t) orelse return fail(req.tag, E.NOENT); // still there? + // Read-only tree, with exactly one exception: the `zmx` file of a + // live agent, and only when the daemon was started to allow moves. + // Truncation is meaningless there (a write replaces the value), so + // it is accepted rather than refused, which is what `>` needs. + const writable = switch (t) { + .act => |a| a.file == .zmx and h.allow_move, + else => false, + }; + const rw = req.omode & 3; + if (!writable) { + if (rw == cloud9.owrite or rw == cloud9.ordwr) return failWhy(req.tag, E.PERM, "9agents serves a view, not a store: writing is not permitted"); + if (req.omode & cloud9.otrunc != 0) return failWhy(req.tag, E.PERM, "truncation is not permitted here"); + } + return .{ .reply = .{ .tag = req.tag, .handle = 1 } }; +} + +fn release(h: *Harness, req: fs.Req, t: Target) Answer { + _ = h; + switch (t) { + .top, .act => {}, + .path => |e| unref(e), + } + return .{ .reply = .{ .tag = req.tag } }; +} + +// ---- reads ---------------------------------------------------------------------- + +fn read(h: *Harness, req: fs.Req, t: Target) Answer { + switch (t) { + .top => |top| switch (top) { + .pid, .uptime, .readme => { + const text = factText(h, top, &fact_buf); + return window(req, text); + }, + else => return fail(req.tag, E.ISDIR), + }, + .act => |a| return activeRead(h, req, a), + .path => |e| { + // `O_NONBLOCK` so a fifo left in a harness root cannot park the + // daemon in `open`; the kind check below refuses it anyway. + const fd = openIn(h, e.root, e.rel(), .{ + .ACCMODE = .RDONLY, + .NONBLOCK = true, + .CLOEXEC = true, + }) orelse return fail(req.tag, E.NOENT); + const file: Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + defer file.close(h.io); + const st = file.stat(h.io) catch return fail(req.tag, E.IO); + if (st.kind == .directory) return fail(req.tag, E.ISDIR); + // Only regular files have bytes this tree promises to serve. + if (st.kind != .file) return fail(req.tag, E.PERM); + _ = linux.fcntl(fd, linux.F.SETFL, 0); // pread wants no O_NONBLOCK + const want = @min(req.size, h.data_buf.len); + const n = file.readPositionalAll(h.io, h.data_buf[0..want], req.off) catch + return fail(req.tag, E.IO); + return .{ .reply = .{ .tag = req.tag }, .bytes = h.data_buf[0..n] }; + }, + } +} + +/// `text` windowed by the request's offset and size. +fn window(req: fs.Req, text: []const u8) Answer { + const off: usize = @intCast(@min(req.off, text.len)); + const n = @min(text.len - off, req.size); + return .{ .reply = .{ .tag = req.tag }, .bytes = text[off..][0..n] }; +} + +// ---- readdir -------------------------------------------------------------------- + +/// Directory records in the engine's shape: `node:u64le dir:u8 len:u8 name`. +const Staging = struct { + buf: []u8, + len: usize = 0, + skip: u64, + /// Set once a record did not fit. Everything after it is left for the + /// next read: dropping one record and staging a shorter one behind it + /// would lose that entry, because the client's next offset counts the + /// records it received. + full: bool = false, + + fn add(s: *Staging, node: u64, dir: bool, name: []const u8) void { + if (s.full) return; + if (s.skip > 0) { + s.skip -= 1; + return; + } + if (name.len == 0 or name.len > 255) return; + if (s.len + 10 + name.len > s.buf.len) { + s.full = true; + return; + } + std.mem.writeInt(u64, s.buf[s.len..][0..8], node, .little); + s.buf[s.len + 8] = @intFromBool(dir); + s.buf[s.len + 9] = @intCast(name.len); + @memcpy(s.buf[s.len + 10 ..][0..name.len], name); + s.len += 10 + name.len; + } +}; + +fn readdir(h: *Harness, req: fs.Req, t: Target) Answer { + var st: Staging = .{ .buf = &h.stage_buf, .skip = req.off }; + switch (t) { + .top => |top| switch (top) { + .root => for (Top.listed) |e| st.add(topNode(e), e.dir(), e.fileName()), + .claude, .codex, .skills => { + for (named_mounts) |m| { + if (m.owner != top) continue; + // Only mounts whose target exists are listed (a harness + // without skills simply has no skills/ entry). + const a = attrOfMount(h, m.mount.root, m.mount.rel) orelse continue; + const e = h.entryFor(m.mount.root, m.mount.rel, false) orelse continue; + st.add(h.entryNode(e), a.dir, m.name); + } + }, + .pid, .uptime, .readme => return fail(req.tag, E.NOTDIR), + .active => { + rescan(h); + for (std.enums.values(active.Kind)) |k| { + if (!hasLive(h, k)) continue; + st.add(activeNode(.{ .file = .harness_dir, .kind = k }), true, k.text()); + } + }, + .omp, .hermes, .dsh => { + const m = top.mirror().?; + if (h.base(m.root) == null) return fail(req.tag, E.NOENT); + switch (listDir(h, &st, m.root, m.rel)) { + .ok => {}, + .gone => return fail(req.tag, E.NOENT), + .failed => return fail(req.tag, E.IO), + .overflow => return failWhy(req.tag, E.NFILE, "directory has more entries than this server can list at once"), + } + }, + }, + .path => |e| { + if (attrFor(h, t)) |a| { + if (!a.dir) return fail(req.tag, E.NOTDIR); + } else return fail(req.tag, E.NOENT); + switch (listDir(h, &st, e.root, e.rel())) { + .ok => {}, + .gone => return fail(req.tag, E.NOENT), + .failed => return fail(req.tag, E.IO), + .overflow => return failWhy(req.tag, E.NFILE, "directory has more entries than this server can list at once"), + } + }, + .act => |a| return activeReaddir(h, req, a), + } + return .{ .reply = .{ .tag = req.tag }, .bytes = h.stage_buf[0..st.len] }; +} + +/// How a listing attempt ended. `overflow` is deliberate: a comptime cap +/// that would silently drop entries answers an error instead, because a +/// short listing is indistinguishable from a small directory. +const Listing = enum { ok, gone, failed, overflow }; + +/// Stages the contents of the directory (root, dir_rel): every name the +/// walker yields that survives the exclusions, sorted so a listing that +/// spans several reads stays consistent. +fn listDir(h: *Harness, st: *Staging, r: Root, dir_rel: []const u8) Listing { + const fd = openIn(h, r, dir_rel, .{ + .ACCMODE = .RDONLY, + .DIRECTORY = true, + .CLOEXEC = true, + }) orelse return .gone; + const dir: Io.Dir = .{ .handle = fd }; + defer Io.Dir.close(dir, h.io); + var read_buf: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var reader = Io.Dir.Reader.init(dir, &read_buf); + var names_len: usize = 0; + var count: usize = 0; + while (true) { + const entry = (reader.next(h.io) catch return .failed) orelse break; + if (excluded(entry.name)) continue; + var kind = entry.kind; + if (kind == .unknown or kind == .sym_link) { + // The walker's word is not proof: stat without following. + var child_buf: [rel_capacity + 256]u8 = undefined; + const child_rel = joinRel(&child_buf, dir_rel, entry.name) orelse continue; + const cst = statIn(h, r, child_rel) orelse continue; + kind = cst.st.kind; + } + if (kind == .sym_link) continue; // never served + // A cap reached is an error, never a short listing: a directory + // that quietly loses entries is a wrong answer, and a caller + // cannot tell it from a small directory. + if (count == list_capacity or names_len + entry.name.len > h.list_names.len) return .overflow; + @memcpy(h.list_names[names_len..][0..entry.name.len], entry.name); + h.list_offs[count] = @intCast(names_len); + h.list_dirs[count] = kind == .directory; + names_len += entry.name.len; + count += 1; + } + // Sort the (offset, length) pairs by name for cursor stability. + const SortCtx = struct { + names: []const u8, + offs: []const u32, + lens: [list_capacity]u32, + + fn lessThan(ctx: @This(), a: usize, b: usize) bool { + return std.mem.order(u8, ctx.nameAt(a), ctx.nameAt(b)) == .lt; + } + fn nameAt(ctx: @This(), i: usize) []const u8 { + const start = ctx.offs[i]; + return ctx.names[start..][0..ctx.lens[i]]; + } + }; + var lens: [list_capacity]u32 = @splat(0); + var order: [list_capacity]usize = @splat(0); + { + var end: usize = 0; + for (0..count) |i| { + end = if (i + 1 < count) h.list_offs[i + 1] else names_len; + lens[i] = @intCast(end - h.list_offs[i]); + order[i] = i; + } + } + const ctx: SortCtx = .{ .names = h.list_names[0..names_len], .offs = &h.list_offs, .lens = lens }; + std.mem.sort(usize, order[0..count], ctx, SortCtx.lessThan); + for (order[0..count]) |i| { + const name = ctx.nameAt(i); + var child_buf: [rel_capacity + 256]u8 = undefined; + const child_rel = joinRel(&child_buf, dir_rel, name) orelse continue; + const e = h.entryFor(r, child_rel, false) orelse return .overflow; // path table full + st.add(h.entryNode(e), h.list_dirs[i], name); + } + return .ok; +} + +// ---- unit tests ------------------------------------------------------------------- + +const testing = std.testing; + +/// A rig with a fake HOME: every root under one temp dir, never the real +/// ~/.claude or any other live harness root. +const Rig = struct { + dir: testing.TmpDir, + /// Build a fixture process tree under /proc and point + /// `/active` at it. Off by default: a unit test must opt in to + /// reading a process tree, and it is never the live one. + with_proc: bool = false, + /// The pid the daemon believes it is, for the ancestry exclusion. + self_pid: u32 = 4242, + zmx: []const u8 = "zmx", + allow_move: bool = false, + path_buf: [std.fs.max_path_bytes]u8 = undefined, + home: []const u8 = undefined, + h: *Harness = undefined, + harness_mem: Harness = undefined, + + fn start(rig: *Rig) !void { + const io = testing.io; + rig.dir = testing.tmpDir(.{}); + errdefer rig.dir.cleanup(); + const len = try rig.dir.dir.realPath(io, &rig.path_buf); + rig.home = try testing.allocator.dupe(u8, rig.path_buf[0..len]); + var mk: [std.fs.max_path_bytes]u8 = undefined; + // The five roots, pinned to the fake home. + inline for ([_][]const u8{ + ".claude/projects/p1", ".claude/skills/revu", + ".codex/sessions/2026/09/21", + ".omp/agent", ".hermes/logs", + ".dsh/profiles", + }) |sub| { + try Io.Dir.cwd().createDirPath(io, try std.fmt.bufPrint(&mk, "{s}/{s}", .{ rig.home, sub })); + } + var base_buf: [5][std.fs.max_path_bytes]u8 = @splat(@splat(0)); + var bases: [5][]const u8 = @splat(""); + inline for (0..5) |i| { + bases[i] = try std.fmt.bufPrint(&base_buf[i], "{s}/{s}", .{ rig.home, home_dirs[i] }); + } + rig.harness_mem = undefined; + var proc_buf: [std.fs.max_path_bytes]u8 = undefined; + const proc_root: []const u8 = if (rig.with_proc) + try std.fmt.bufPrint(&proc_buf, "{s}/proc", .{rig.home}) + else + ""; + if (rig.with_proc) { + try Io.Dir.cwd().createDirPath(io, proc_root); + try rig.put("proc/stat", "cpu 1 2 3\nbtime 1000000\nprocesses 7\n"); + } + rig.harness_mem.init(.{ + .io = io, + .pid = rig.self_pid, + .bases = bases, + .proc = proc_root, + .home = rig.home, + .zmx = rig.zmx, + .allow_move = rig.allow_move, + }); + rig.h = &rig.harness_mem; + } + + fn end(rig: *Rig) void { + rig.dir.cleanup(); + testing.allocator.free(rig.home); + } + + fn put(rig: *Rig, rel_path: []const u8, bytes: []const u8) !void { + var buf: [std.fs.max_path_bytes]u8 = undefined; + const path = try std.fmt.bufPrint(&buf, "{s}/{s}", .{ rig.home, rel_path }); + if (std.mem.lastIndexOfScalar(u8, rel_path, '/')) |i| { + var dbuf: [std.fs.max_path_bytes]u8 = undefined; + const dir_path = try std.fmt.bufPrint(&dbuf, "{s}/{s}", .{ rig.home, rel_path[0..i] }); + try Io.Dir.cwd().createDirPath(testing.io, dir_path); + } + var file = try Io.Dir.createFileAbsolute(testing.io, path, .{}); + defer file.close(testing.io); + try file.writeStreamingAll(testing.io, bytes); + } + + + fn del(rig: *Rig, rel_path: []const u8) !void { + var buf: [std.fs.max_path_bytes]u8 = undefined; + const path = try std.fmt.bufPrint(&buf, "{s}/{s}", .{ rig.home, rel_path }); + try Io.Dir.deleteFileAbsolute(testing.io, path); + } + + /// lookup of `name` in `dir_node`, answering the child's attr. + fn lookupName(rig: *Rig, tag: u64, dir_node: u64, name: []const u8) Answer { + return handle(rig.h, .{ .tag = tag, .op = .lookup, .node = dir_node, .data = name }); + } + + /// Writes one fake process into the fixture `/proc`: the `stat` + /// line (with the start time in field 22, after a name that holds + /// the spaces and parentheses a real one can), the NUL-separated + /// `cmdline`, and a `cwd` symlink. + fn fakeProc(rig: *Rig, o: struct { + pid: u32, + ppid: u32 = 1, + comm: []const u8 = "x", + starttime: u64 = 5000, + argv: []const u8, + cwd: []const u8 = "", + }) !void { + var rel: [128]u8 = undefined; + var stat_text: [512]u8 = undefined; + var w = Io.Writer.fixed(&stat_text); + try w.print("{d} ({s}) S {d}", .{ o.pid, o.comm, o.ppid }); + for (5..22) |i| try w.print(" {d}", .{i}); + try w.print(" {d} 0 0\n", .{o.starttime}); + try rig.put(try std.fmt.bufPrint(&rel, "proc/{d}/stat", .{o.pid}), w.buffered()); + try rig.put(try std.fmt.bufPrint(&rel, "proc/{d}/cmdline", .{o.pid}), o.argv); + + const target = if (o.cwd.len > 0) o.cwd else rig.home; + var link_buf: [std.fs.max_path_bytes]u8 = undefined; + const link = try std.fmt.bufPrintZ(&link_buf, "{s}/proc/{d}/cwd", .{ rig.home, o.pid }); + Io.Dir.cwd().symLink(testing.io, target, link, .{}) catch {}; + } + + /// Removes a fake process, as an exit would. + fn reapProc(rig: *Rig, pid: u32) !void { + var buf: [std.fs.max_path_bytes]u8 = undefined; + const dir_path = try std.fmt.bufPrint(&buf, "{s}/proc/{d}", .{ rig.home, pid }); + const d = try Io.Dir.openDirAbsolute(testing.io, dir_path, .{ .iterate = true }); + var rb: [Io.Dir.Iterator.reader_buffer_len]u8 align(@alignOf(usize)) = undefined; + var it = Io.Dir.Reader.init(d, &rb); + var names: [8][64]u8 = undefined; + var lens: [8]usize = @splat(0); + var n: usize = 0; + while (n < names.len) { + const e = (it.next(testing.io) catch break) orelse break; + @memcpy(names[n][0..e.name.len], e.name); + lens[n] = e.name.len; + n += 1; + } + Io.Dir.close(d, testing.io); + for (0..n) |i| { + var one: [std.fs.max_path_bytes]u8 = undefined; + const path = try std.fmt.bufPrint(&one, "{s}/{s}", .{ dir_path, names[i][0..lens[i]] }); + Io.Dir.deleteFileAbsolute(testing.io, path) catch {}; + } + try Io.Dir.cwd().deleteDir(testing.io, dir_path); + } +}; + +const home_dirs = [5][]const u8{ ".claude", ".codex", ".omp", ".hermes", ".dsh" }; + +fn expectNoent(a: Answer) !void { + try testing.expect(a.reply.status == .err); + try testing.expectEqual(E.NOENT, a.reply.errno); +} + +fn expectEperm(a: Answer) !void { + try testing.expect(a.reply.status == .err); + try testing.expectEqual(E.PERM, a.reply.errno); +} + +test "tree: facts at the root" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + const a = rig.lookupName(1, root, "pid"); + try testing.expect(a.reply.status == .ok); + try testing.expect(a.reply.attr.dir == false); + var read_a = handle(rig.h, .{ .tag = 2, .op = .read, .node = a.reply.attr.node, .off = 0, .size = 64 }); + try testing.expectEqualStrings("4242\n", read_a.bytes); + read_a = handle(rig.h, .{ .tag = 3, .op = .read, .node = a.reply.attr.node, .off = 0, .size = 2 }); + try testing.expectEqualStrings("42", read_a.bytes); + const up = rig.lookupName(4, root, "uptime"); + try testing.expect(up.reply.status == .ok); + const up_read = handle(rig.h, .{ .tag = 5, .op = .read, .node = up.reply.attr.node, .off = 0, .size = 64 }); + try testing.expect(up_read.bytes.len > 0 and up_read.bytes[up_read.bytes.len - 1] == '\n'); + // The root lists the facts and the five harness dirs. + const listing = handle(rig.h, .{ .tag = 6, .op = .readdir, .node = root, .off = 0, .size = msize }); + try testing.expect(listing.reply.status == .ok); + for ([_][]const u8{ "README", "pid", "uptime", "claude", "codex", "omp", "hermes", "dsh", "skills" }) |name| { + try testing.expect(stageHas(listing.bytes, name)); + } +} + +test "tree: README is served at the root, whole and read-only" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + + const rd = rig.lookupName(1, root, "README"); + try testing.expect(rd.reply.status == .ok); + try testing.expect(!rd.reply.attr.dir); + try testing.expectEqual(readme_text.len, rd.reply.attr.size); + + // Read it back whole and compare: a truncated or empty README would + // otherwise pass every check above. + const got = handle(rig.h, .{ .tag = 2, .op = .read, .node = rd.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(got.reply.status == .ok); + try testing.expectEqualStrings(readme_text, got.bytes); + + // It is a file, and it is read-only like everything else here. + const as_dir = handle(rig.h, .{ .tag = 3, .op = .readdir, .node = rd.reply.attr.node, .off = 0, .size = msize }); + try testing.expectEqual(E.NOTDIR, as_dir.reply.errno); + const wr = handle(rig.h, .{ .tag = 4, .op = .write, .node = rd.reply.attr.node, .off = 0, .size = 1 }); + try testing.expectEqual(E.PERM, wr.reply.errno); +} + +fn stageHas(staged: []const u8, name: []const u8) bool { + var i: usize = 0; + while (i + 10 <= staged.len) { + const len: usize = staged[i + 9]; + const entry = staged[i + 10 ..][0..len]; + if (std.mem.eql(u8, entry, name)) return true; + i += 10 + len; + } + return false; +} + +test "tree: exclusions never serve a credentials-shaped name" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + // The fixture: a real transcript and a pile of credentials-shaped names + // at several depths, including inside the mirrored subtrees. + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"a\":1}\n"); + try rig.put(".claude/projects/p1/.credentials.json", "STAY OUT\n"); + try rig.put(".claude/projects/p1/auth.json", "STAY OUT\n"); + try rig.put(".claude/projects/p1/settings.json", "STAY OUT\n"); + try rig.put(".claude/projects/p1/token.txt", "STAY OUT\n"); + try rig.put(".claude/projects/p1/api.key", "STAY OUT\n"); + try rig.put(".claude/projects/p1/models.yml", "STAY OUT\n"); + try rig.put(".claude/projects/p1/deep/.env", "STAY OUT\n"); + try rig.put(".claude/history.jsonl", "{\"h\":1}\n"); + try rig.put(".hermes/auth.json", "STAY OUT\n"); + try rig.put(".hermes/.env", "STAY OUT\n"); + try rig.put(".hermes/config.yaml", "STAY OUT\n"); + try rig.put(".hermes/logs/app.log", "log line\n"); + try rig.put(".dsh/.credentials.yaml", "STAY OUT\n"); + try rig.put(".dsh/settings.yaml", "STAY OUT\n"); + + // /claude lists only its mounts; the credentials at ~/.claude root are + // not in the tree at all (no mount serves them). + const claude = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.claude), .off = 0, .size = msize }); + try testing.expect(claude.reply.status == .ok); + for ([_][]const u8{ "projects", "history", "skills" }) |name| try testing.expect(stageHas(claude.bytes, name)); + try testing.expect(!stageHas(claude.bytes, "credentials")); + + // A project's listing shows the transcript and nothing else. + const projects = rig.lookupName(2, topNode(.claude), "projects"); + try testing.expect(projects.reply.status == .ok); + const p1 = rig.lookupName(3, projects.reply.attr.node, "p1"); + try testing.expect(p1.reply.status == .ok); + const listed = handle(rig.h, .{ .tag = 4, .op = .readdir, .node = p1.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(listed.reply.status == .ok); + try testing.expect(stageHas(listed.bytes, "session-x.jsonl")); + for ([_][]const u8{ ".credentials.json", "auth.json", "settings.json", "token.txt", "api.key", "models.yml" }) |name| { + try testing.expect(!stageHas(listed.bytes, name)); + } + // The plain directory that holds a .env is served; the .env is not. + try testing.expect(stageHas(listed.bytes, "deep")); + + // Every credentials-shaped name is unreachable by lookup too. + for ([_][]const u8{ ".credentials.json", "auth.json", "settings.json", "token.txt", "api.key", "models.yml" }) |name| { + try expectNoent(rig.lookupName(5, p1.reply.attr.node, name)); + } + const deep = rig.lookupName(6, p1.reply.attr.node, "deep"); + try testing.expect(deep.reply.status == .ok); + try expectNoent(rig.lookupName(7, deep.reply.attr.node, ".env")); + + // The fully mirrored roots hide theirs as well. + const hermes = rig.lookupName(8, root, "hermes"); + try testing.expect(hermes.reply.status == .ok); + const hlist = handle(rig.h, .{ .tag = 9, .op = .readdir, .node = hermes.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(hlist.bytes, "logs")); + for ([_][]const u8{ "auth.json", ".env", "config.yaml" }) |name| { + try testing.expect(!stageHas(hlist.bytes, name)); + try expectNoent(rig.lookupName(10, hermes.reply.attr.node, name)); + } + const dsh = rig.lookupName(11, root, "dsh"); + try testing.expect(dsh.reply.status == .ok); + const dlist = handle(rig.h, .{ .tag = 12, .op = .readdir, .node = dsh.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(!stageHas(dlist.bytes, ".credentials.yaml")); + try testing.expect(!stageHas(dlist.bytes, "settings.yaml")); + try testing.expect(stageHas(dlist.bytes, "profiles")); + + // The exclusion predicate itself, spelled out. + try testing.expect(excluded(".credentials.json")); + try testing.expect(excluded("AUTH.JSON")); + try testing.expect(excluded("session-token.bin")); + try testing.expect(excluded("id_rsa_backup")); + try testing.expect(excluded("prod.pem")); + try testing.expect(excluded("config.yaml")); + try testing.expect(!excluded("session-x.jsonl")); + try testing.expect(!excluded("SKILL.md")); + try testing.expect(!excluded("logs")); +} + +test "tree: paths compose only from the pinned roots" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + // A slash, a NUL or an overlong name never reaches the filesystem. + try expectNoent(rig.lookupName(1, topNode(.claude), "projects/../p1")); + try expectNoent(rig.lookupName(2, topNode(.claude), "projects/\x00")); + // A symlink inside a mirror is not served, whatever it points at. + var buf: [std.fs.max_path_bytes]u8 = undefined; + const link = try std.fmt.bufPrintZ(&buf, "{s}/.claude/projects/p1/escape", .{rig.home}); + try Io.Dir.cwd().symLink(testing.io, rig.home, link, .{}); + const projects = rig.lookupName(3, topNode(.claude), "projects"); + const p1 = rig.lookupName(4, projects.reply.attr.node, "p1"); + try expectNoent(rig.lookupName(5, p1.reply.attr.node, "escape")); + // ".." from a mirrored file lands on its canonical parent, and walking + // ".." repeatedly terminates at the root. + const session = rig.lookupName(6, p1.reply.attr.node, "session-x.jsonl"); + try testing.expect(session.reply.status == .ok); + var cur = session.reply.attr.node; + var hops: usize = 0; + while (hops < 8) : (hops += 1) { + const up = rig.lookupName(7, cur, ".."); + try testing.expect(up.reply.status == .ok); + if (up.reply.attr.node == root) break; + cur = up.reply.attr.node; + } + try testing.expect(cur == root or hops < 8); + // The facts refuse lookups with NOTDIR. + try testing.expect(rig.lookupName(8, topNode(.pid), "x").reply.errno == E.NOTDIR); +} + +test "tree: a file's qid path survives release, and its version tracks content" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const dir = rig.lookupName(2, projects.reply.attr.node, "p1"); + const first = rig.lookupName(3, dir.reply.attr.node, "session-x.jsonl"); + try testing.expect(first.reply.status == .ok); + const qid_path = first.reply.attr.path.?; + const version = first.reply.attr.version; + + // Give every reference back, so the table slot is free to recycle. + for ([_]u64{ first.reply.attr.node, dir.reply.attr.node, projects.reply.attr.node }) |n| { + _ = handle(rig.h, .{ .tag = 4, .op = .release, .node = n }); + } + + // Walk to it again from scratch. intro(5): two files are the same file if + // and only if their qids are the same — so the same unchanged file must + // come back with the same qid path, whatever the node id does. Before the + // qid path came from the file itself, the recycled slot's new generation + // made this a different file on every walk. + const p2 = rig.lookupName(5, topNode(.claude), "projects"); + const d2 = rig.lookupName(6, p2.reply.attr.node, "p1"); + const again = rig.lookupName(7, d2.reply.attr.node, "session-x.jsonl"); + try testing.expect(again.reply.status == .ok); + try testing.expectEqual(qid_path, again.reply.attr.path.?); + try testing.expectEqual(version, again.reply.attr.version); + + // Two different files never share a qid path. + try rig.put(".claude/projects/p1/session-y.jsonl", "{}\n"); + const other = rig.lookupName(8, d2.reply.attr.node, "session-y.jsonl"); + try testing.expect(other.reply.attr.path.? != qid_path); + + // And a change to the content moves the version, which is what tells a + // caching client the transcript it is holding has grown. + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n{\"more\":1}\n"); + const grown = rig.lookupName(9, d2.reply.attr.node, "session-x.jsonl"); + try testing.expectEqual(qid_path, grown.reply.attr.path.?); + try testing.expect(grown.reply.attr.version != version); +} + +test "tree: node ids — same file equal, distinct files differ, ids recycle" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + try rig.put(".claude/projects/p1/session-y.jsonl", "{}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const a1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + // Two walks of the same file: the same node id (the dedupe table). + const x1 = rig.lookupName(3, a1.reply.attr.node, "session-x.jsonl"); + const x2 = rig.lookupName(4, a1.reply.attr.node, "session-x.jsonl"); + try testing.expect(x1.reply.status == .ok); + try testing.expectEqual(x1.reply.attr.node, x2.reply.attr.node); + const y1 = rig.lookupName(5, a1.reply.attr.node, "session-y.jsonl"); + try testing.expect(y1.reply.attr.node != x1.reply.attr.node); + // The union skills view and the harness's own skills share identity. + const via_harness = rig.lookupName(6, topNode(.claude), "skills"); + const via_union = rig.lookupName(7, topNode(.skills), "claude"); + try testing.expectEqual(via_harness.reply.attr.node, via_union.reply.attr.node); + // References are paid back by release: both slots go, ids recycle. + for ([_]u64{ x1.reply.attr.node, x2.reply.attr.node, y1.reply.attr.node, a1.reply.attr.node }) |n| { + const rel = handle(rig.h, .{ .tag = 8, .op = .release, .node = n }); + try testing.expect(rel.reply.status == .ok); + } + const x3 = rig.lookupName(9, projects.reply.attr.node, "p1"); + const x4 = rig.lookupName(10, x3.reply.attr.node, "session-x.jsonl"); + // The entry was freed and re-created: a fresh generation, a new id. + try testing.expect(x4.reply.attr.node != x1.reply.attr.node); + // Node packing round-trips through the struct. + const n: Node = .{ .idx = 3, .kind = 1, .serial = 0x1234_5678_9abc }; + const bits: u64 = @bitCast(n); + const back: Node = @bitCast(bits); + try testing.expect(back.idx == 3 and back.kind == 1 and back.serial == 0x1234_5678_9abc); +} + +test "tree: a file that vanishes between lookup and read answers ENOENT" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"a\":1}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); + try testing.expect(session.reply.status == .ok); + // The transcript reads back whole, fresh from disk. + const whole = handle(rig.h, .{ .tag = 4, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 1024 }); + try testing.expectEqualStrings("{\"a\":1}\n", whole.bytes); + // It vanishes: lookup, getattr and read all answer ENOENT cleanly. + try rig.del(".claude/projects/p1/session-x.jsonl"); + try expectNoent(rig.lookupName(5, p1.reply.attr.node, "session-x.jsonl")); + const gone = handle(rig.h, .{ .tag = 6, .op = .getattr, .node = session.reply.attr.node }); + try expectNoent(gone); + const read_gone = handle(rig.h, .{ .tag = 7, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 64 }); + try expectNoent(read_gone); +} + +test "tree: writes and setattrs answer EPERM" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); + try testing.expect(session.reply.status == .ok); + const write = handle(rig.h, .{ .tag = 4, .op = .write, .node = session.reply.attr.node, .data = "x" }); + try expectEperm(write); + const set = handle(rig.h, .{ .tag = 5, .op = .setattr, .node = session.reply.attr.node, .set = .{ .mtime = true }, .mtime = 1 }); + try expectEperm(set); + // An open for write is refused at open time. + const wopen = handle(rig.h, .{ .tag = 6, .op = .open, .node = session.reply.attr.node, .omode = cloud9.owrite }); + try expectEperm(wopen); + const ropen = handle(rig.h, .{ .tag = 7, .op = .open, .node = session.reply.attr.node, .omode = cloud9.oread }); + try testing.expect(ropen.reply.status == .ok); +} + +test "tree: a transcript written while serving is visible at once" { + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"n\":1}\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + // The harness appends; the next read sees it — no cache in between. + try rig.put(".claude/projects/p1/session-x.jsonl", "{\"n\":1}\n{\"n\":2}\n"); + const session = rig.lookupName(3, p1.reply.attr.node, "session-x.jsonl"); + const fresh = handle(rig.h, .{ .tag = 4, .op = .read, .node = session.reply.attr.node, .off = 0, .size = 1024 }); + try testing.expectEqualStrings("{\"n\":1}\n{\"n\":2}\n", fresh.bytes); + // A brand-new session file appears on the next listing. + try rig.put(".claude/projects/p1/session-new.jsonl", "{\"n\":9}\n"); + const listed = handle(rig.h, .{ .tag = 5, .op = .readdir, .node = p1.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(listed.bytes, "session-new.jsonl")); +} + +comptime { + // Every static mount target must survive the exclusion rules; the + // daemon's own skeleton may never be filtered out from under it. + for (named_mounts) |m| { + if (excludedPath(m.mount.rel)) @compileError("a 9agents mount target is excluded by name"); + } +} + +fn stageCount(staged: []const u8) usize { + var i: usize = 0; + var n: usize = 0; + while (i + 10 <= staged.len) : (n += 1) i += 10 + @as(usize, staged[i + 9]); + return n; +} + +test "tree: a file directly under a mirror root reads back" { + // Regression: joining a child onto an empty relative path returned an + // uncopied scratch slice, so every file at the top of a mirror root + // (/hermes/, /dsh/) listed but resolved to garbage bytes. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put(".hermes/note.txt", "hello-hermes\n"); + const listing = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.hermes), .off = 0, .size = msize }); + try testing.expect(stageHas(listing.bytes, "note.txt")); + const note = rig.lookupName(2, topNode(.hermes), "note.txt"); + try testing.expect(note.reply.status == .ok); + const bytes = handle(rig.h, .{ .tag = 3, .op = .read, .node = note.reply.attr.node, .off = 0, .size = 64 }); + try testing.expectEqualStrings("hello-hermes\n", bytes.bytes); +} + +test "tree: a name swapped for a symlink under an open handle serves nothing" { + // Regression: the read path composed the whole path and opened it in + // one call, following symlinks. A name replaced between the walk and + // the read handed the client bytes from outside every pinned root. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + try rig.put("outside.txt", "OUTSIDE-THE-ROOTS\n"); // beside the roots, not in one + try rig.put(".claude/projects/p1/swap.txt", "safe\n"); + const projects = rig.lookupName(1, topNode(.claude), "projects"); + const p1 = rig.lookupName(2, projects.reply.attr.node, "p1"); + const swap = rig.lookupName(3, p1.reply.attr.node, "swap.txt"); + try testing.expect(swap.reply.status == .ok); + const opened = handle(rig.h, .{ .tag = 4, .op = .open, .node = swap.reply.attr.node, .omode = cloud9.oread }); + try testing.expect(opened.reply.status == .ok); + // The file becomes a symlink out of the tree while the handle is open. + var link_buf: [std.fs.max_path_bytes]u8 = undefined; + var target_buf: [std.fs.max_path_bytes]u8 = undefined; + const link = try std.fmt.bufPrintZ(&link_buf, "{s}/.claude/projects/p1/swap.txt", .{rig.home}); + const target = try std.fmt.bufPrint(&target_buf, "{s}/outside.txt", .{rig.home}); + try rig.del(".claude/projects/p1/swap.txt"); + try Io.Dir.cwd().symLink(testing.io, target, link, .{}); + const after = handle(rig.h, .{ .tag = 5, .op = .read, .node = swap.reply.attr.node, .off = 0, .size = 64 }); + try expectNoent(after); + try testing.expectEqual(@as(usize, 0), after.bytes.len); +} + +test "tree: a listing past the cap fails loudly instead of truncating" { + // Regression: a directory with more entries (or longer names) than the + // comptime caps was served short, and a short listing is + // indistinguishable from a small directory. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + var name_buf: [64]u8 = undefined; + for (0..list_capacity + 1) |i| { + try rig.put(try std.fmt.bufPrint(&name_buf, ".dsh/f{d:0>5}", .{i}), ""); + } + const listed = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.dsh), .off = 0, .size = msize }); + try testing.expect(listed.reply.status == .err); + try testing.expectEqual(E.NFILE, listed.reply.errno); +} + +test "tree: a listing spanning several reads loses no entry" { + // The staging buffer fills long before a big directory ends. Staging + // must stop at the first record that does not fit: dropping it and + // packing a shorter one behind it would lose that entry, because the + // next read's offset counts the records already delivered. + var rig: Rig = .{ .dir = undefined }; + try rig.start(); + defer rig.end(); + const count = 200; + var name_buf: [320]u8 = undefined; + for (0..count) |i| { + const name = try std.fmt.bufPrint(&name_buf, ".dsh/{d:0>3}{s}", .{ i, "n" ** 200 }); + try rig.put(name, ""); + } + var seen: [count]bool = @splat(false); + var off: u64 = 0; + var reads: usize = 0; + while (reads < 16) : (reads += 1) { + const page = handle(rig.h, .{ .tag = 1, .op = .readdir, .node = topNode(.dsh), .off = off, .size = msize }); + try testing.expect(page.reply.status == .ok); + if (page.bytes.len == 0) break; + var i: usize = 0; + while (i + 10 <= page.bytes.len) { + const len: usize = page.bytes[i + 9]; + const name = page.bytes[i + 10 ..][0..len]; + i += 10 + len; + // The fixture root also holds `profiles`, created by the rig. + const idx = std.fmt.parseInt(usize, name[0..@min(3, name.len)], 10) catch continue; + try testing.expect(!seen[idx]); // never served twice + seen[idx] = true; + } + off += stageCount(page.bytes); + } + try testing.expect(reads > 1); // the listing really did span several reads + for (seen) |s| try testing.expect(s); +} + +test { + _ = active; // the derived view's own tests run with the tree's +} + +// ---- /active: the derived view ------------------------------------------------ + +/// Walks `/active` down to one agent's directory, answering its node. +fn activeEntryNode(rig: *Rig, harness: []const u8, pid: []const u8) !u64 { + const act = rig.lookupName(90, root, "active"); + try testing.expect(act.reply.status == .ok); + // A listing is what rescans, so it comes before the walk. + _ = handle(rig.h, .{ .tag = 91, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + const h_dir = rig.lookupName(92, act.reply.attr.node, harness); + try testing.expect(h_dir.reply.status == .ok); + const entry = rig.lookupName(93, h_dir.reply.attr.node, pid); + try testing.expect(entry.reply.status == .ok); + return entry.reply.attr.node; +} + +fn activeField(rig: *Rig, entry: u64, name: []const u8) ![]const u8 { + const f = rig.lookupName(94, entry, name); + try testing.expect(f.reply.status == .ok); + const r = handle(rig.h, .{ .tag = 95, .op = .read, .node = f.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(r.reply.status == .ok); + return r.bytes; +} + +test "active: a fixture process tree lists by harness and by pid" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + + // A Claude Code session, resolved the way the harness publishes it. + try rig.fakeProc(.{ .pid = 1001, .comm = "claude", .starttime = 5000, .argv = "claude\x00--print\x00" }); + var rec: [512]u8 = undefined; + try rig.put(".claude/sessions/1001.json", try std.fmt.bufPrint(&rec, + \\{{"pid":1001,"sessionId":"sess-abc","cwd":"{s}", + \\ "name":"fixture-one","status":"busy","procStart":"5000"}} + , .{rig.home})); + var slug_buf: [512]u8 = undefined; + const slug = active.slugOf(.claude, rig.home, rig.home, &slug_buf).?; + var tr: [640]u8 = undefined; + try rig.put( + try std.fmt.bufPrint(&tr, ".claude/projects/{s}/sess-abc.jsonl", .{slug}), + "{\"type\":\"summary\",\"aiTitle\":\"Fixture Title\"}\n", + ); + + const entry = try activeEntryNode(&rig, "claude", "1001"); + try testing.expectEqualStrings("1001\n", try activeField(&rig, entry, "pid")); + try testing.expectEqualStrings("sess-abc\n", try activeField(&rig, entry, "session")); + try testing.expectEqualStrings("registry\n", try activeField(&rig, entry, "via")); + try testing.expectEqualStrings("fixture-one\n", try activeField(&rig, entry, "name")); + try testing.expectEqualStrings("busy\n", try activeField(&rig, entry, "status")); + try testing.expectEqualStrings("Fixture Title\n", try activeField(&rig, entry, "title")); + // started is the boot time plus the process's own, in seconds. + try testing.expectEqualStrings("1000050\n", try activeField(&rig, entry, "started")); + // The transcript is served as the file it is, not as a field. + const t = rig.lookupName(96, entry, "transcript"); + try testing.expect(t.reply.status == .ok); + const bytes = handle(rig.h, .{ .tag = 97, .op = .read, .node = t.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(std.mem.indexOf(u8, bytes.bytes, "Fixture Title") != null); + // A field this harness does not publish is absent, not blank. + try expectNoent(rig.lookupName(98, entry, "model")); +} + +test "active: a daemon or helper is never listed as an agent" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + // The shapes actually running on this machine. + try rig.fakeProc(.{ .pid = 1010, .comm = "python3", .argv = "python3\x00-m\x00hermes_cli.main\x00gateway\x00run\x00" }); + try rig.fakeProc(.{ .pid = 1011, .comm = "omp", .argv = "omp\x00__omp_worker_daemon_broker\x00" }); + try rig.fakeProc(.{ .pid = 1012, .comm = "claude", .argv = "claude\x00mcp-server\x00" }); + // ...and one real session, so an empty answer cannot pass by default. + try rig.fakeProc(.{ .pid = 1013, .comm = "claude", .argv = "claude\x00" }); + + const act = rig.lookupName(1, root, "active"); + const listing = handle(rig.h, .{ .tag = 2, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(listing.bytes, "claude")); + try testing.expect(!stageHas(listing.bytes, "hermes")); + try testing.expect(!stageHas(listing.bytes, "omp")); + const claude = rig.lookupName(3, act.reply.attr.node, "claude"); + const pids = handle(rig.h, .{ .tag = 4, .op = .readdir, .node = claude.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(pids.bytes, "1013")); + try testing.expect(!stageHas(pids.bytes, "1012")); // the mcp server +} + +test "active: a pid reused by another process stops resolving" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + try rig.fakeProc(.{ .pid = 1020, .comm = "claude", .starttime = 5000, .argv = "claude\x00" }); + const entry = try activeEntryNode(&rig, "claude", "1020"); + try testing.expect(handle(rig.h, .{ .tag = 5, .op = .getattr, .node = entry }).reply.status == .ok); + + // The same pid, a different process: only the start time says so. + try rig.fakeProc(.{ .pid = 1020, .comm = "claude", .starttime = 9999, .argv = "claude\x00" }); + const act = rig.lookupName(6, root, "active"); + _ = handle(rig.h, .{ .tag = 7, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + try expectNoent(handle(rig.h, .{ .tag = 8, .op = .getattr, .node = entry })); + // The pid is still there — as the new process, under a new node. + const fresh = try activeEntryNode(&rig, "claude", "1020"); + try testing.expect(fresh != entry); +} + +test "active: a process that exits leaves the tree" { + var rig: Rig = .{ .dir = undefined, .with_proc = true }; + try rig.start(); + defer rig.end(); + try rig.fakeProc(.{ .pid = 1030, .comm = "claude", .argv = "claude\x00" }); + const entry = try activeEntryNode(&rig, "claude", "1030"); + try testing.expect(handle(rig.h, .{ .tag = 9, .op = .getattr, .node = entry }).reply.status == .ok); + + try rig.reapProc(1030); + const act = rig.lookupName(10, root, "active"); + const listing = handle(rig.h, .{ .tag = 11, .op = .readdir, .node = act.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(!stageHas(listing.bytes, "claude")); + try expectNoent(handle(rig.h, .{ .tag = 12, .op = .getattr, .node = entry })); + try expectNoent(rig.lookupName(13, act.reply.attr.node, "claude")); +} + +test "active: the daemon never lists the process tree it lives in" { + // 1041 is the daemon, its parent 1040 is a harness: showing it would + // let a client act on the tree serving it. + var rig: Rig = .{ .dir = undefined, .with_proc = true, .self_pid = 1041 }; + try rig.start(); + defer rig.end(); + try rig.fakeProc(.{ .pid = 1040, .comm = "claude", .argv = "claude\x00" }); + try rig.fakeProc(.{ .pid = 1041, .comm = "9agents", .ppid = 1040, .argv = "9agents\x00" }); + try rig.fakeProc(.{ .pid = 1042, .comm = "claude", .argv = "claude\x00" }); + + const act = rig.lookupName(14, root, "active"); + const claude = rig.lookupName(15, act.reply.attr.node, "claude"); + try testing.expect(claude.reply.status == .ok); + const pids = handle(rig.h, .{ .tag = 16, .op = .readdir, .node = claude.reply.attr.node, .off = 0, .size = msize }); + try testing.expect(stageHas(pids.bytes, "1042")); // an unrelated session lists + try testing.expect(!stageHas(pids.bytes, "1040")); // its own parent does not +} diff --git a/9agents/test/e2e.sh b/9agents/test/e2e.sh new file mode 100755 index 0000000..fa41bef --- /dev/null +++ b/9agents/test/e2e.sh @@ -0,0 +1,358 @@ +#!/usr/bin/env bash +# End-to-end suite for 9agents: the read-only, fresh-from-disk 9P view of +# every harness's state. +# +# Usage: bash 9agents/test/e2e.sh <9agents> [<9ns>] (zig build 9agents-itest) +# +# Part A always runs, entirely on fixtures: a fake home with fixture roots +# pinned by --root, a scratch XDG_RUNTIME_DIR registry, plan9port's 9p as +# the client. It proves: the posted name is listed, the roots walk and +# read, a transcript comes back byte-identical (cmp), credentials-shaped +# files are unreachable, a file appended after the daemon started is +# visible at once, writes answer EPERM, and --unix/--fd listen forms work. +# +# Part B (the money shot) runs against the REAL roots and the REAL +# registry only when the `agents` name is free, and only reads: the posted +# name in /run/user//9p, 9p walks of the live ~/.claude, a real +# transcript byte-identical through the tree, the 9ns --mntgen mount of +# /mnt/9p/agents, and the live ~/.claude/.credentials.json unreachable. +# It is skipped (not failed) when a live daemon already owns the name. +# +# Exit 0 on success (or when the machine cannot run a part), 1 on failure. +set -u + +H9=$(realpath "${1:?path to 9agents}") +HARNESS_TMP_XDG="${XDG_RUNTIME_DIR:-}" +[ $# -ge 2 ] && [ -n "$2" ] && NS=$(realpath "$2") +P9P=/usr/lib/plan9/bin/9p +TMP=$(mktemp -d "${TMPDIR:-/tmp}/9agents.XXXXXX") +PIDS=() +FAILED=0 +PASSED=0 + +cleanup() { + for p in "${PIDS[@]:-}"; do [ -n "$p" ] && kill "$p" 2>/dev/null; done + rm -rf "$TMP" + # part B posts into the *real* registry under a scratch name; a crash + # between the post and the unpost would otherwise leave a socket there. + [ -n "${REAL_SOCKET:-}" ] && rm -f "$REAL_SOCKET" + return 0 +} +trap cleanup EXIT + +pass() { PASSED=$((PASSED + 1)); echo "ok - $1"; } +fail() { FAILED=$((FAILED + 1)); echo "FAIL - $1"; shift; [ $# -gt 0 ] && printf ' %s\n' "$@"; } +expect_eq() { # name expected actual + if [ "$2" = "$3" ]; then pass "$1"; else fail "$1" "expected: $(printf %q "$2")" "actual: $(printf %q "$3")"; fi +} +expect_contains() { # name needle haystack + case "$3" in *"$2"*) pass "$1" ;; *) fail "$1" "missing: $(printf %q "$2")" "in: $(printf %q "$3")" ;; esac +} +expect_missing() { # name needle haystack + case "$3" in *"$2"*) fail "$1" "found: $(printf %q "$2")" "in: $(printf %q "$3")" ;; *) pass "$1" ;; esac +} + +if [ ! -x "$P9P" ]; then + echo "SKIP: plan9port 9p not at $P9P"; exit 0 +fi +if ! unshare -Urm true 2>/dev/null; then + echo "SKIP: unprivileged user namespaces unavailable"; exit 0 +fi + +start_daemon() { # args... -> sets DAEMON_PID, logs to $TMP/daemon.log + "$H9" "$@" >"$TMP/daemon.log" 2>&1 & + DAEMON_PID=$! + PIDS+=("$DAEMON_PID") +} +wait_posted() { # socket path + for _ in $(seq 1 100); do [ -S "$1" ] && return 0; sleep 0.05; done + return 1 +} +p9() { "$P9P" -a "unix!$1" "${@:2}"; } + +# ============================================================================ +echo "# part A: fixture roots, scratch registry" +# ============================================================================ +export XDG_RUNTIME_DIR="$TMP" +REG="$TMP/9p" +FX="$TMP/home" +mkdir -p "$FX" + +# The fixture home: five roots, a real-shaped claude project with a +# transcript, credentials-shaped files at several depths, codex sessions, +# an omp agent dir, hermes logs and dsh profiles. +mkfile() { mkdir -p "$(dirname "$1")"; printf '%s' "$2" > "$1"; } +mkfile "$FX/claude/projects/-tmp-proj/session-abc.jsonl" '{"type":"user","message":"first line"} +{"type":"assistant","message":"second line"} +' +mkfile "$FX/claude/projects/-tmp-proj/.credentials.json" 'STAY OUT' +mkfile "$FX/claude/projects/-tmp-proj/auth.json" 'STAY OUT' +mkfile "$FX/claude/projects/-tmp-proj/token.bin" 'STAY OUT' +mkfile "$FX/claude/history.jsonl" '{"display":"claude prompt one"} +{"display":"claude prompt two"} +' +mkdir -p "$FX/claude/skills/revu" +mkfile "$FX/claude/skills/revu/SKILL.md" '# revu skill' +mkfile "$FX/codex/sessions/2026/09/21/rollout-x.jsonl" '{"session":"codex one"} +' +mkfile "$FX/codex/session_index.jsonl" '{"id":"x"} +' +mkfile "$FX/codex/history.jsonl" '{"text":"codex prompt"} +' +mkfile "$FX/omp/agent/history.db" 'fake-history-db' +mkfile "$FX/omp/agent/config.yml" 'STAY OUT' +mkfile "$FX/hermes/auth.json" 'STAY OUT' +mkfile "$FX/hermes/logs/app.log" 'hermes log line +' +mkfile "$FX/hermes/state.db" 'fake-hermes-state' +mkfile "$FX/dsh/profiles/p1.yaml" 'name: p1' +mkfile "$FX/dsh/.credentials.yaml" 'STAY OUT' + +start_daemon --root "claude=$FX/claude" --root "codex=$FX/codex" \ + --root "omp=$FX/omp" --root "hermes=$FX/hermes" --root "dsh=$FX/dsh" \ + --name agents +wait_posted "$REG/agents" || { fail "the daemon posts as agents" "$(cat "$TMP/daemon.log")"; exit 1; } + +# 1. The posted name is listed. +expect_contains "posted name listed in the registry" agents "$(ls "$REG")" + +# 2. The roots walk; a real transcript reads back byte-identical. +expect_contains "ls / shows the roots" claude "$(p9 "$REG/agents" ls /)" +expect_contains "ls / shows codex" codex "$(p9 "$REG/agents" ls /)" +expect_contains "ls / shows skills" skills "$(p9 "$REG/agents" ls /)" +p9 "$REG/agents" read /claude/projects/-tmp-proj/session-abc.jsonl > "$TMP/via-9p.jsonl" 2>"$TMP/read.err" \ + || fail "transcript read through the tree" "$(cat "$TMP/read.err")" +cmp -s "$TMP/via-9p.jsonl" "$FX/claude/projects/-tmp-proj/session-abc.jsonl" \ + && pass "transcript byte-identical through the tree (cmp)" \ + || fail "transcript byte-identical through the tree (cmp)" "differs" +expect_eq "history file reads" "$(cat "$FX/claude/history.jsonl")" "$(p9 "$REG/agents" read /claude/history)" +expect_eq "skills union mirrors the harness tree" "$(cat "$FX/claude/skills/revu/SKILL.md")" \ + "$(p9 "$REG/agents" read /skills/claude/revu/SKILL.md)" + +# 3. A credentials-shaped file is unreachable anywhere in the tree. +PROJ_LS=$(p9 "$REG/agents" ls /claude/projects/-tmp-proj) +for bad in .credentials.json auth.json token.bin; do + expect_missing "credentials-shaped $bad not listed" "$bad" "$PROJ_LS" + p9 "$REG/agents" read "/claude/projects/-tmp-proj/$bad" >/dev/null 2>&1 \ + && fail "credentials-shaped $bad unreachable" "read succeeded" \ + || pass "credentials-shaped $bad unreachable" +done +expect_missing "hermes auth.json not listed" auth.json "$(p9 "$REG/agents" ls /hermes)" +expect_missing "omp config.yml not listed" config.yml "$(p9 "$REG/agents" ls /omp)" +expect_missing "dsh .credentials.yaml not listed" credentials.yaml "$(p9 "$REG/agents" ls /dsh)" +sleep 0.3 +expect_contains "hermes logs still served" logs "$(p9 "$REG/agents" ls /hermes)" + +# 3b. A file at the very top of a mirror root: its relative path is the +# bare name, the one case a join onto an empty directory path gets wrong. +expect_contains "a file at the top of a mirror root is listed" state.db "$(p9 "$REG/agents" ls /hermes)" +expect_eq "a file at the top of a mirror root reads back" "$(cat "$FX/hermes/state.db")" \ + "$(p9 "$REG/agents" read /hermes/state.db)" + +# 3c. A directory bigger than the listing caps fails loudly. A short +# listing is indistinguishable from a small directory, so the daemon must +# never answer one: the read errors and the client sees it. +mkdir -p "$FX/dsh/wide" +seq 1 1100 | while read -r i; do : > "$FX/dsh/wide/f$(printf %05d "$i")"; done +if p9 "$REG/agents" ls /dsh/wide > "$TMP/wide.out" 2>"$TMP/wide.err"; then + fail "an over-cap directory fails instead of truncating" "listed $(wc -l < "$TMP/wide.out") entries" +else + pass "an over-cap directory fails instead of truncating ($(head -c 80 "$TMP/wide.err"))" +fi +rm -rf "$FX/dsh/wide" + +# 4. Live visibility: a file appended after the daemon started. +printf '{"type":"assistant","message":"third line"}\n' >> "$FX/claude/projects/-tmp-proj/session-abc.jsonl" +expect_eq "append after start is visible immediately" \ + "$(cat "$FX/claude/projects/-tmp-proj/session-abc.jsonl")" \ + "$(p9 "$REG/agents" read /claude/projects/-tmp-proj/session-abc.jsonl)" +mkdir -p "$FX/claude/projects/-tmp-proj2" +mkfile "$FX/claude/projects/-tmp-proj2/session-new.jsonl" '{"new":true} +' +expect_contains "new session dir appears at once" -tmp-proj2 "$(p9 "$REG/agents" ls /claude/projects)" + +# 5. The facts and the write refusal. +DAEMON_PID_TEXT=$(p9 "$REG/agents" read /pid) +expect_eq "pid fact answers the daemon's pid" "$DAEMON_PID" "$DAEMON_PID_TEXT" +[ -n "$(p9 "$REG/agents" read /uptime)" ] && pass "uptime fact answers" || fail "uptime fact answers" "empty" +printf 'x' | p9 "$REG/agents" write /pid >/dev/null 2>&1 \ + && fail "write answers EPERM" "write succeeded" \ + || pass "write answers EPERM" + +# 6. The --unix listen form beside the post. +"$H9" --unix "$TMP/plain.sock" --no-post --root "claude=$FX/claude" >"$TMP/d2.log" 2>&1 & +PIDS+=($!) +for _ in $(seq 1 100); do [ -S "$TMP/plain.sock" ] && break; sleep 0.05; done +expect_eq "--unix listen form serves" "$(cat "$FX/claude/history.jsonl")" "$(p9 "$TMP/plain.sock" read /claude/history)" + +# 7. The --fd listen form: one 9P session over a connected stream fd +# (the 9ns --spawn / socket-activation shape), driven through a +# socketpair: the daemon sees EOF when both ends close and exits. +if command -v python3 >/dev/null; then + python3 - "$H9" "$FX" <<'EOF' +import os, socket, subprocess, sys +h9, fx = sys.argv[1], sys.argv[2] +a, b = socket.socketpair() +pid = os.fork() +if pid == 0: + a.close() + os.dup2(b.fileno(), 7) + os.execv(h9, [h9, "--fd", "7", "--root", f"claude={fx}/claude"]) +b.close() +a.close() +os.waitpid(pid, 0) +EOF + pass "--fd listen form runs a session and exits on hangup" +else + echo "skip - --fd form: python3 not available" +fi + +kill "$DAEMON_PID" 2>/dev/null +wait "$DAEMON_PID" 2>/dev/null +sleep 0.2 +[ -S "$REG/agents" ] && fail "SIGTERM unposts the name" "socket still there" || pass "SIGTERM unposts the name" + +# ============================================================================ +echo "# part C: /active, the derived view" +# ============================================================================ +# A fake agent: a process whose argv[0] basename is a harness name, with a +# session record Claude Code's own shape. The proc root is the real /proc +# (the fixture seam is unit-tested); nothing real is ever killed, because +# the only process this part touches is the sleep it started itself. +ACT="$TMP/act" +mkdir -p "$ACT/bin" "$ACT/home/.claude/sessions" "$ACT/home/.claude/projects" "$ACT/rt/9p/zmx" +cp /bin/sleep "$ACT/bin/claude" +# A zmx that records how it was called and posts the name, as the real one +# does; a move must never be tested against the user's live sessions. +cat > "$ACT/bin/zmx" < "$ACT/zmx-argv" +: > "$ACT/rt/9p/zmx/\$2" +exit 0 +ZEOF +chmod +x "$ACT/bin/zmx" + +"$ACT/bin/claude" 600 & +AGENT=$! +PIDS+=("$AGENT") +sleep 0.3 +# Field 22 of /proc//stat, counted from the last ')' — the executable +# name can hold spaces and parentheses, so the line is never just split. +AGENT_START=$(sed 's/.*) //' "/proc/$AGENT/stat" | awk '{print $20}') +AGENT_CWD=$(readlink "/proc/$AGENT/cwd") +printf '{"pid":%d,"sessionId":"sess-e2e","cwd":"%s","name":"e2e-agent","status":"idle","procStart":"%s"}\n' \ + "$AGENT" "$AGENT_CWD" "$AGENT_START" > "$ACT/home/.claude/sessions/$AGENT.json" + +act_roots() { + echo --root "claude=$ACT/home/.claude" --root "codex=$ACT/home/.codex" \ + --root "omp=$ACT/home/.omp" --root "hermes=$ACT/home/.hermes" --root "dsh=$ACT/home/.dsh" +} + +# First: the read-only default. No --allow-move, so zmx cannot be written. +XDG_RUNTIME_DIR="$ACT/rt" "$H9" --no-post --unix "$ACT/ro.sock" $(act_roots) >"$ACT/ro.log" 2>&1 & +PIDS+=("$!") +wait_posted "$ACT/ro.sock" || { fail "read-only daemon listens" "$(cat "$ACT/ro.log")"; } +expect_contains "the agent lists under its harness" "$AGENT" "$(p9 "$ACT/ro.sock" ls /active/claude)" +expect_eq "its session comes from the harness's own record" "sess-e2e" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/session")" +expect_eq "the route taken is named" "registry" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/via")" +expect_eq "the harness's own name is served" "e2e-agent" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/name")" +expect_eq "the harness's own status is served" "idle" "$(p9 "$ACT/ro.sock" read "/active/claude/$AGENT/status")" +if echo "nope" | p9 "$ACT/ro.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null; then + fail "without --allow-move the tree stays read-only" "the write was accepted" +else + pass "without --allow-move the tree stays read-only" +fi +expect_eq "a refused move leaves the agent running" "yes" "$([ -d "/proc/$AGENT" ] && echo yes)" + +# Now a daemon that allows moves. +XDG_RUNTIME_DIR="$ACT/rt" "$H9" --no-post --unix "$ACT/rw.sock" $(act_roots) \ + --allow-move --zmx "$ACT/bin/zmx" >"$ACT/rw.log" 2>&1 & +PIDS+=("$!") +wait_posted "$ACT/rw.sock" || { fail "move-allowing daemon listens" "$(cat "$ACT/rw.log")"; } + +# Every refusal must come before anything is destroyed. +for bad in "has space" "../escape" "semi;colon"; do + echo "$bad" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null + if [ $? -eq 0 ]; then fail "an illegal zmx name is refused ($bad)"; else pass "an illegal zmx name is refused ($bad)"; fi +done +: > "$ACT/rt/9p/zmx/occupied" +echo "occupied" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>/dev/null \ + && fail "a name a live session holds is refused" || pass "a name a live session holds is refused" +expect_eq "no refusal killed the agent" "yes" "$([ -d "/proc/$AGENT" ] && echo yes)" + +# The move itself. +if echo "e2e-moved" | p9 "$ACT/rw.sock" write "/active/claude/$AGENT/zmx" 2>"$ACT/move.err"; then + pass "a legal name moves the agent" +else + fail "a legal name moves the agent" "$(cat "$ACT/move.err")" +fi +sleep 0.4 +expect_eq "the agent it replaced is gone" "gone" "$([ -d "/proc/$AGENT" ] || echo gone)" +# The command is fixed by the harness: the client supplied only the name. +expect_eq "zmx ran the harness with its session resumed" \ + "run e2e-moved -d claude --resume sess-e2e" "$(cat "$ACT/zmx-argv")" +expect_missing "no client byte reached the command" "e2e-moved -d claude --resume sess-e2e;" "$(cat "$ACT/zmx-argv")" + +# ============================================================================ +echo "# part B: the real roots, the real registry (read-only)" +# ============================================================================ +export XDG_RUNTIME_DIR="${HARNESS_TMP_XDG:-/run/user/$(id -u)}" +REAL_REG="$XDG_RUNTIME_DIR/9p" +# A scratch name, not `agents`: the whole point of part B is to read the real +# roots through the real registry, and that has nothing to do with which name +# the tree is posted under. Taking `agents` would mean this suite could only +# run while the machine's own 9agents.service was stopped — so it would either +# be skipped on any machine that actually uses the daemon, or fight it. +REAL_NAME="agents-itest-$$" +REAL_SOCKET="$REAL_REG/$REAL_NAME" +if [ -e "$REAL_SOCKET" ]; then + # $$ collided with a leftover socket from a crashed run of this suite. + echo "SKIP: scratch name $REAL_NAME is already taken" +else + HOME_DIR=$(getent passwd "$(id -u)" | cut -d: -f6) + start_daemon --name "$REAL_NAME" + if wait_posted "$REAL_SOCKET"; then + expect_contains "posted name listed in the real registry" "$REAL_NAME" "$(ls "$REAL_REG")" + expect_contains "ls / shows the claude root" claude "$(p9 "$REAL_SOCKET" ls /)" + # A real transcript, byte-identical through the tree. + REAL_T=$(ls "$HOME_DIR/.claude/projects"/*/*.jsonl 2>/dev/null | head -n 1 || true) + if [ -n "$REAL_T" ]; then + REL="${REAL_T#"$HOME_DIR/.claude/projects/"}" + p9 "$REAL_SOCKET" read "/claude/projects/$REL" > "$TMP/real-via-9p" 2>/dev/null \ + && cmp -s "$TMP/real-via-9p" "$REAL_T" \ + && pass "real transcript byte-identical through the tree" \ + || fail "real transcript byte-identical through the tree" "$REAL_T" + else + echo "skip - no real claude transcript found" + fi + # The credentials at ~/.claude are unreachable: no mount serves the + # root dir, and the exclusion rule holds everywhere else. + p9 "$REAL_SOCKET" read /claude/.credentials.json >/dev/null 2>&1 \ + && fail "live .credentials.json unreachable" "read succeeded" \ + || pass "live .credentials.json unreachable" + CLAUDE_LS=$(p9 "$REAL_SOCKET" ls /claude) + expect_missing "no credentials leaked into /claude" credentials "$CLAUDE_LS" + # The money shot: the mntgen mount every interactive fish sees. + if [ -n "$NS" ]; then + OUT=$(timeout 60 "$NS" --mntgen -- sh -c "ls /mnt/9p/$REAL_NAME && head -c 200 /mnt/9p/$REAL_NAME/claude/history" 2>"$TMP/mntgen.err") + RC=$? + if [ $RC -eq 0 ]; then + expect_contains "mntgen mount lists the agents tree" claude "$OUT" + expect_contains "mntgen mount reads claude/history" claude "$OUT" + else + fail "mntgen money shot (exit $RC)" "$(cat "$TMP/mntgen.err")" + fi + else + echo "skip - mntgen money shot: 9ns not provided" + fi + kill "$DAEMON_PID" 2>/dev/null + wait "$DAEMON_PID" 2>/dev/null + sleep 0.2 + [ -S "$REAL_SOCKET" ] && fail "stop unposts the real name" "socket still there" || pass "stop unposts the real name" + else + fail "daemon posts into the real registry" "$(cat "$TMP/daemon.log")" + fi +fi + +echo "# $PASSED passed, $FAILED failed" +[ "$FAILED" -eq 0 ] -- cgit v1.3