//! The chat behind `/active///chat/`: a harness transcript //! read as numbered messages, one file each. //! //! chat/00000000-user 00000001-assistant 00000002-bash 00000003-bash-result ... //! //! The name is `-`. The id is the message's position in the //! transcript, counted from 0 and written as 8 digits so the names sort //! in order; transcripts are append-only, so an id never moves and a new //! message takes the next one. The kind says who spoke, normalized across //! harnesses (`Kind`); a tool call is named after its tool and its result //! after the call (`Chat.label`). The file holds the message rendered as //! text: JSON strings unescaped, and each part on its own line — a tool //! call is its name, then its input. //! //! A transcript is JSONL, and finding message N means knowing where every //! message before it starts, so each transcript gets an index: per message, //! the spans of the file that render it (`Span`). The index is not a copy //! of anything. It holds offsets, never bytes, and every read still reads //! the file. It is keyed by the file's (dev, ino) and extended from where it //! stopped when the file grows; a file that shrinks or is replaced starts //! over. Only complete lines are indexed, so a record the harness is //! half-way through writing waits for its newline. //! //! The caps are loud, as everywhere else in 9agents: a transcript with more //! messages than an index holds is marked `full`, and the listing refuses //! rather than end early. const std = @import("std"); const Io = std.Io; const linux = std.os.linux; // ---- comptime bounds --------------------------------------------------------- /// Messages one index holds. The longest codex rollout on this machine /// (49k records) is 32k messages; Claude Code sessions stay near 3k. The /// pool is mapped `MAP_NORESERVE` (`Pool.create`), so a generous cap costs /// address space, not memory: a page is only backed once an index uses it. pub const max_msgs: usize = 1 << 18; /// Spans one index holds: most messages are one span, a tool call two. pub const max_spans: usize = 2 * max_msgs; /// Digits in a message id: every id has the same width, so a plain sort /// of the names is the order the messages were said in. pub const id_digits: usize = 8; comptime { std.debug.assert(max_msgs <= 100_000_000); // every id fits the width } /// Transcripts indexed at once, reused least-recently-used first. pub const chat_slots: usize = 8; /// The read window over a transcript. pub const window: usize = 64 * 1024; /// The window always shows at least this much past the position asked for, /// unless the file ends first: enough for any key compared and for the /// longest JSON escape (a surrogate pair, 12 bytes). const lookahead: usize = 64; /// The transcript formats this module reads. pub const Format = enum(u8) { claude, codex }; /// Who spoke. The name of a message file ends in `text()`. pub const Kind = enum(u8) { /// A person's prompt. user, /// The model's reply text. assistant, /// The model's reasoning, where the harness writes it in the clear. thinking, /// A tool the model called: its name, then its input. Named after /// the tool when it has a usable name; `tool-call` when not. tool_call, /// What the tool answered. Named `-result` when its call is /// known; `tool-result` when not. tool_result, /// The harness speaking: developer instructions, slash-command /// output, injected context, compaction summaries. system, /// Another agent of a team (codex): its name, then what it said. agent, pub fn text(k: Kind) []const u8 { return switch (k) { .tool_call => "tool-call", .tool_result => "tool-result", else => @tagName(k), }; } pub fn fromText(s: []const u8) ?Kind { for (std.enums.values(Kind)) |k| if (std.mem.eql(u8, k.text(), s)) return k; return null; } }; /// How a span's bytes become text. pub const Enc = enum(u2) { /// The bytes as they are (a JSON value served verbatim). raw, /// The body of a JSON string, unescaped. str, /// A fixed text from `literals`; `off` is its index. lit, }; /// Texts for what a transcript holds that is not text. const literals = [_][]const u8{ "[image]", "" }; const lit_image: u64 = 0; const lit_empty: u64 = 1; /// One stretch of the transcript that renders part of a message. /// Packed to 16 bytes: an index is mostly these. pub const Span = packed struct(u128) { /// File offset of the first byte; for `lit`, an index into `literals`. off: u48, /// Bytes in the file. len: u32, /// Bytes it renders to, not counting `nl`. out: u32, enc: Enc, /// A newline follows, because the text does not end in one. nl: bool, _: u13 = 0, fn size(s: Span) u64 { return @as(u64, s.out) + @intFromBool(s.nl); } }; pub const Msg = struct { /// Index of the first span in `Chat.spans`. first: u32, count: u16, kind: Kind, /// Unix seconds of the record, 0 when it carries no timestamp. time: u32, /// Rendered size: the sum of the spans'. size: u32, /// For a tool call or result, the tool, as an index into the chat's /// tool names plus one; 0 when unknown. tool: u8 = 0, }; /// Distinct tool names one chat names its calls by. Past this, a call is /// named `tool-call`, as one whose tool has no usable name is. pub const max_tools: usize = 255; /// Longest tool name kept for a file name. pub const tool_name_cap: usize = 48; /// Calls remembered for matching their results: results follow their /// calls closely, a few dozen at most when calls run in parallel. const call_ring: usize = 256; const Call = struct { /// Hash of the call's id (never 0 when set). id: u64 = 0, tool: u8 = 0, }; /// The index of one transcript. pub const Chat = struct { used: bool = false, format: Format = .claude, dev: u64 = 0, ino: u64 = 0, /// The file's size when last synced; reads never go past it. size: u64 = 0, /// Offset of the first line not yet indexed. Always a line start. scanned: u64 = 0, /// Least-recently-used ordering. stamp: u64 = 0, /// A cap was reached. Messages past `n` exist but are not indexed. full: bool = false, /// When the file was created, where the filesystem says: a new file /// that reuses a deleted one's inode is born later. btime: i128 = 0, /// Hashes of the first and last bytes indexed (up to `probe_len` each, /// the last ending at `scanned`). A transcript rewritten in place to /// the same size or longer keeps its inode and grows past `scanned`; /// only its bytes can say it is no longer the file that was indexed. head: u64 = 0, tail: u64 = 0, /// Bumped by every reset, so a read cursor cannot outlive its index. epoch: u32 = 0, /// Offset up to which the unindexed tail is known to hold no newline. unterminated: u64 = 0, n: u32 = 0, nspans: u32 = 0, ntools: u8 = 0, call_next: u16 = 0, tool_lens: [max_tools]u8, tool_names: [max_tools][tool_name_cap]u8, calls: [call_ring]Call, msgs: [max_msgs]Msg, spans: [max_spans]Span, fn reset(c: *Chat, format: Format, dev: u64, ino: u64) void { c.used = true; c.epoch +%= 1; c.btime = 0; c.head = 0; c.tail = 0; c.format = format; c.dev = dev; c.ino = ino; c.size = 0; c.scanned = 0; c.full = false; c.unterminated = 0; c.n = 0; c.nspans = 0; c.ntools = 0; c.call_next = 0; c.calls = @splat(.{}); } pub fn msg(c: *const Chat, i: u32) ?Msg { return if (i < c.n) c.msgs[i] else null; } /// What a message's name says after its id: its kind — or, for a /// tool call, the tool, and for its result the tool and `-result`. fn label(c: *const Chat, m: Msg) struct { []const u8, []const u8 } { if (m.tool == 0) return .{ m.kind.text(), "" }; const t = c.tool_names[m.tool - 1][0..c.tool_lens[m.tool - 1]]; return .{ t, if (m.kind == .tool_result) "-result" else "" }; } /// `-