diff options
| author | Gabriel Schneider <[email protected]> | 2026-07-30 17:05:12 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-07-30 17:56:23 -0300 |
| commit | 8ec4e135e318c7799b2576a8729493c063da4a97 (patch) | |
| tree | ab37cdbb7f478eb996264cf941891f127f303c85 /vrsite/transcript.go | |
| parent | cb05c363045bf0e7af5858f6d7bc26f026fa9d70 (diff) | |
| download | notevi-8ec4e135e318c7799b2576a8729493c063da4a97.tar.gz notevi-8ec4e135e318c7799b2576a8729493c063da4a97.zip | |
Diffstat (limited to 'vrsite/transcript.go')
| -rw-r--r-- | vrsite/transcript.go | 602 |
1 files changed, 602 insertions, 0 deletions
diff --git a/vrsite/transcript.go b/vrsite/transcript.go new file mode 100644 index 0000000..713fe36 --- /dev/null +++ b/vrsite/transcript.go @@ -0,0 +1,602 @@ +// transcript.go — the conversation a session's reading happened inside. +// +// The vr log says a session read a file at 10:04; the harness that ran that +// session kept the whole conversation on disk, in its own format. Here we find +// that file by session id and flatten both known formats into Turns, so a note +// can be opened at the moment it was written. +// +// Two rules keep this cheap and safe. Tool calls and their output are reduced +// to a line each — transcripts reach megabytes and none of that belongs in a +// page — and nothing outside a configured root is ever opened, with session +// ids checked before they touch the filesystem. +package main + +import ( + "encoding/json" + "fmt" + "os" + "path/filepath" + "sort" + "strings" + "sync" + "time" +) + +const ( + maxTranscript = 32 << 20 // refuse to parse more than this in one file + transcriptCacheMax = 8 // parsed transcripts are large; keep few + briefLine = 160 // one line of a tool call or its output + maxText = 6000 // one message + convWindow = 80 // turns rendered around the anchored moment + convMax = 1500 // turns rendered by "show all" +) + +// A Turn is one moment in a session: something said, a tool called or answered, +// or — woven in from the vr log — a read, a grep, a note. +type Turn struct { + Role string // user | assistant | thinking | tool | result | vr + Kind string // qualifies Role: the vr op, or "error" on a failed tool + When time.Time + Text string // what was said + Tool string // tool name, on tool and result turns + Detail string // one compact line: the call's arguments, or its output + File string // vr turns: the file the op touched + Note *Note // vr note turns: the note itself, rendered as a card + Here bool // the anchored moment +} + +func (t Turn) Clock() string { + if t.When.IsZero() { + return "" + } + return t.When.Local().Format("15:04:05") +} + +// A Transcript is one harness conversation, already normalized. +type Transcript struct { + Path string + Harness string // claude-code | codex + Turns []Turn + Err string +} + +// transcripts finds and parses harness transcripts, remembering what it has +// already parsed. Roots are the only places it will look. +type transcripts struct { + roots []string + + mu sync.Mutex + cache map[string]*Transcript // path|mtime|size -> parsed +} + +func newTranscripts(roots []string) *transcripts { + tx := &transcripts{cache: map[string]*Transcript{}} + for _, r := range roots { + r = strings.TrimSpace(r) + if r == "" { + continue + } + if abs, err := filepath.Abs(r); err == nil { + r = abs + } + tx.roots = append(tx.roots, filepath.Clean(r)) + } + return tx +} + +// defaultTranscriptRoots are where the two harnesses we know about keep their +// conversations. +func defaultTranscriptRoots() []string { + home, err := os.UserHomeDir() + if err != nil { + return nil + } + return []string{filepath.Join(home, ".claude", "projects"), filepath.Join(home, ".codex", "sessions")} +} + +// safeSession refuses ids that could name something we did not mean to open: +// the id goes into a filesystem path, so it must be one plain component. +func safeSession(id string) error { + if id == "" { + return fmt.Errorf("no session id") + } + if len(id) > 128 { + return fmt.Errorf("session id is too long") + } + for _, r := range id { + ok := r == '-' || r == '_' || r == ':' || r == '.' || + '0' <= r && r <= '9' || 'a' <= r && r <= 'z' || 'A' <= r && r <= 'Z' + if !ok { + return fmt.Errorf("session id contains %q", r) + } + } + if strings.Contains(id, "..") { + return fmt.Errorf("session id contains %q", "..") + } + return nil +} + +// mangleDir is how Claude Code names a project directory after its cwd: +// /home/u/0x4200.cafe -> -home-u-0x4200-cafe. +func mangleDir(dir string) string { + return strings.NewReplacer("/", "-", ".", "-").Replace(dir) +} + +// under states the rule the globs below already obey: a transcript we open +// lies inside a configured root. It is checked anyway, so the rule has one +// place to fail rather than living in the shape of two patterns. +func under(root, path string) bool { + rel, err := filepath.Rel(root, path) + if err != nil { + return false + } + return rel != ".." && !strings.HasPrefix(rel, ".."+string(filepath.Separator)) +} + +// find locates a session's transcript. Claude Code writes +// <root>/<mangled cwd>/<session>.jsonl; codex writes +// <root>/YYYY/MM/DD/rollout-<time>-<thread>.jsonl and vr's session id is that +// thread id, so it is matched on the suffix. Both patterns are tried under +// every root — a root knows its own layout, we do not have to. +func (tx *transcripts) find(session, dir string) string { + if safeSession(session) != nil { + return "" + } + for _, root := range tx.roots { + pats := []string{ + filepath.Join(root, "*", session+".jsonl"), + filepath.Join(root, "*", "*", "*", "rollout-*-"+session+".jsonl"), + } + // the cwd we logged names the directory directly, when it still exists + if dir != "" { + pats = append([]string{filepath.Join(root, mangleDir(dir), session+".jsonl")}, pats...) + } + for _, pat := range pats { + hits, _ := filepath.Glob(pat) + for _, hit := range hits { + if under(root, hit) { + return hit + } + } + } + } + return "" +} + +// load returns the parsed transcript for a session, or nil when there is none. +// Parses are cached by path, mtime and size, so a 5 MB file is read once. +func (tx *transcripts) load(session, dir string) *Transcript { + path := tx.find(session, dir) + if path == "" { + return nil + } + st, err := os.Stat(path) + if err != nil { + return nil + } + key := fmt.Sprintf("%s|%d|%d", path, st.ModTime().UnixNano(), st.Size()) + + tx.mu.Lock() + t, ok := tx.cache[key] + tx.mu.Unlock() + if ok { + return t + } + + t = readTranscript(path, st.Size()) + tx.mu.Lock() + if len(tx.cache) >= transcriptCacheMax { + tx.cache = map[string]*Transcript{} + } + tx.cache[key] = t + tx.mu.Unlock() + return t +} + +// ── parsing ─────────────────────────────────────────────────────────────── + +// One line of either format. Codex wraps every record in "payload"; Claude +// Code puts the model turn in "message". Sniffing per line keeps the two +// parsers independent of any file header. +type rawRecord struct { + Type string `json:"type"` + Timestamp string `json:"timestamp"` + Payload json.RawMessage `json:"payload"` + Message json.RawMessage `json:"message"` +} + +type parser struct { + t *Transcript + tools map[string]string // tool call id -> name, so results keep their name + last time.Time // records without a stamp inherit the previous one +} + +func readTranscript(path string, size int64) *Transcript { + t := &Transcript{Path: path} + if size > maxTranscript { + t.Err = fmt.Sprintf("transcript is %d MB — too large to parse", size>>20) + return t + } + b, err := os.ReadFile(path) + if err != nil { + t.Err = err.Error() + return t + } + p := &parser{t: t, tools: map[string]string{}} + for _, line := range strings.Split(string(b), "\n") { + if line == "" { + continue + } + var rec rawRecord + if json.Unmarshal([]byte(line), &rec) != nil { + continue + } + at := parseTime(rec.Timestamp) + if at.IsZero() { + at = p.last + } else { + p.last = at + } + switch { + case len(rec.Payload) > 0: + t.Harness = "codex" + p.codex(rec, at) + case len(rec.Message) > 0: + t.Harness = "claude-code" + p.claude(rec, at) + } + } + return t +} + +// add drops turns that would render as nothing — empty thinking blocks, +// tool results whose output was pure whitespace. +func (p *parser) add(t Turn) { + if t.Text == "" && t.Detail == "" && t.Tool == "" { + return + } + t.Text = clip(t.Text, maxText) + p.t.Turns = append(p.t.Turns, t) +} + +// ── claude code ─────────────────────────────────────────────────────────── + +type claudeMsg struct { + Role string `json:"role"` + Content json.RawMessage `json:"content"` // a string, or blocks +} + +type claudeBlock struct { + Type string `json:"type"` + Text string `json:"text"` + Thinking string `json:"thinking"` + ID string `json:"id"` + Name string `json:"name"` + Input json.RawMessage `json:"input"` + ToolUseID string `json:"tool_use_id"` + Content json.RawMessage `json:"content"` // tool_result: a string, or blocks + IsError bool `json:"is_error"` +} + +// claude turns records of type user and assistant into Turns; the rest of the +// file (modes, titles, file history, attachments) is harness bookkeeping. +func (p *parser) claude(rec rawRecord, at time.Time) { + if rec.Type != "user" && rec.Type != "assistant" { + return + } + var m claudeMsg + if json.Unmarshal(rec.Message, &m) != nil { + return + } + var text string + if json.Unmarshal(m.Content, &text) == nil { // plain prompts arrive as a string + p.add(Turn{Role: rec.Type, When: at, Text: text}) + return + } + var blocks []claudeBlock + if json.Unmarshal(m.Content, &blocks) != nil { + return + } + for _, b := range blocks { + switch b.Type { + case "text": + p.add(Turn{Role: rec.Type, When: at, Text: b.Text}) + case "thinking": + p.add(Turn{Role: "thinking", When: at, Text: b.Thinking}) + case "tool_use": + p.tools[b.ID] = b.Name + p.add(Turn{Role: "tool", When: at, Tool: b.Name, Detail: callDetail(b.Input)}) + case "tool_result": + t := Turn{Role: "result", When: at, Tool: p.tools[b.ToolUseID], Detail: resultDetail(b.Content)} + if b.IsError { + t.Kind = "error" + } + p.add(t) + } + } +} + +// ── codex ───────────────────────────────────────────────────────────────── + +type codexPayload struct { + Type string `json:"type"` + Role string `json:"role"` + Name string `json:"name"` + CallID string `json:"call_id"` + Arguments json.RawMessage `json:"arguments"` // a JSON string holding JSON + Input json.RawMessage `json:"input"` + Output json.RawMessage `json:"output"` + Content []codexText `json:"content"` + Summary []codexText `json:"summary"` +} + +type codexText struct { + Type string `json:"type"` + Text string `json:"text"` +} + +// codex reads response_items — the model's own history. The event_msg stream +// beside it is the TUI's view of the same turns, so taking both would double +// everything. +func (p *parser) codex(rec rawRecord, at time.Time) { + if rec.Type != "response_item" { + return + } + var pl codexPayload + if json.Unmarshal(rec.Payload, &pl) != nil { + return + } + switch pl.Type { + case "message": + // developer messages are the harness's own instructions, and the first + // user messages are context it injects; neither was said by anyone + if pl.Role == "developer" { + return + } + text := joinText(pl.Content) + if injected(text) { + return + } + role := "assistant" + if pl.Role == "user" { + role = "user" + } + p.add(Turn{Role: role, When: at, Text: text}) + case "reasoning": + // summaries are usually encrypted and come back empty; add drops those + p.add(Turn{Role: "thinking", When: at, Text: joinText(pl.Summary)}) + case "function_call", "custom_tool_call", "web_search_call": + name := pl.Name + if name == "" { + name = strings.TrimSuffix(pl.Type, "_call") + } + p.tools[pl.CallID] = name + detail := callDetail(pl.Arguments) + if detail == "" { + detail = callDetail(pl.Input) + } + p.add(Turn{Role: "tool", When: at, Tool: name, Detail: detail}) + case "function_call_output", "custom_tool_call_output", "web_search_output": + p.add(Turn{Role: "result", When: at, Tool: p.tools[pl.CallID], Detail: resultDetail(pl.Output)}) + } +} + +func joinText(items []codexText) string { + var parts []string + for _, it := range items { + if it.Text != "" { + parts = append(parts, it.Text) + } + } + return strings.Join(parts, "\n\n") +} + +// injected recognizes the wrappers a harness puts around a user turn to carry +// its own state. +func injected(text string) bool { + for _, tag := range []string{"<environment_context>", "<user_instructions>", "<permissions instructions>"} { + if strings.HasPrefix(text, tag) { + return true + } + } + return false +} + +// ── summarizing ─────────────────────────────────────────────────────────── + +// callFields are the arguments worth a line, most telling first. Anything that +// carries file content (new_string, content) is deliberately not here. +var callFields = []string{"command", "cmd", "file_path", "path", "pattern", "query", "url", "prompt", "description"} + +// callDetail reduces a tool call's arguments to their one interesting field. +func callDetail(raw json.RawMessage) string { + if len(raw) == 0 { + return "" + } + var s string + if json.Unmarshal(raw, &s) == nil { // codex passes arguments as a JSON string + raw = json.RawMessage(s) + } + var m map[string]any + if json.Unmarshal(raw, &m) != nil { + return brief(string(raw)) + } + for _, k := range callFields { + if v, ok := m[k].(string); ok && strings.TrimSpace(v) != "" { + return brief(v) + } + } + b, err := json.Marshal(m) + if err != nil { + return "" + } + return brief(string(b)) +} + +// resultDetail summarizes what a tool answered, whether that came back as a +// string, as content blocks, or as an image nobody wants inlined. +func resultDetail(raw json.RawMessage) string { + if len(raw) == 0 { + return "" + } + var s string + if json.Unmarshal(raw, &s) == nil { + return brief(s) + } + var items []codexText + if json.Unmarshal(raw, &items) == nil { + var parts []string + for _, it := range items { + switch { + case it.Text != "": + parts = append(parts, it.Text) + case it.Type != "": + parts = append(parts, "["+it.Type+"]") + } + } + return brief(strings.Join(parts, "\n")) + } + return brief(string(raw)) +} + +// brief keeps a blob's first line and says how much was left behind. Whole +// files and command output pass through here; a page gets their shape, never +// their bytes. +func brief(s string) string { + s = strings.TrimRight(s, "\n") + if strings.TrimSpace(s) == "" { + return "" + } + first, rest, more := strings.Cut(s, "\n") + first = clip(strings.TrimSpace(first), briefLine) + if !more || strings.TrimSpace(rest) == "" { + return first + } + return fmt.Sprintf("%s … +%d lines", first, strings.Count(rest, "\n")+1) +} + +func clip(s string, n int) string { + if len(s) <= n { // bytes first: the common case never allocates + return s + } + r := []rune(s) + if len(r) <= n { + return s + } + return strings.TrimRight(string(r[:n]), " \t") + "…" +} + +// ── one session: the transcript with the vr log woven in ────────────────── + +// A Conv is what the session page shows: the conversation, the log entries +// that happened during it, and where in a long one we are looking. +type Conv struct { + Session, Agent, Model, Dir string + Bit uint8 + Harness, Path string + Roots []string + Err string + Reads, Greps, Notes int + + Turns []Turn + Total int // moments in the whole conversation + From, To int // 1-based position of Turns within it + Anchor string // element id to scroll to, "" when nothing is anchored + AllHref string // same page, unwindowed + Windowed bool + Incomplete bool // even "show all" had to stop +} + +// sessionTurns are the vr log's own moments for a session: what it read, +// grepped and noted. +func (s *Site) sessionTurns(id string) []Turn { + var out []Turn + for _, e := range s.Entries { + if e.Session != id || e.Op == "note" { + continue + } + out = append(out, Turn{Role: "vr", Kind: e.Op, When: parseTime(e.Time), + Text: describe(e), File: e.File}) + } + for _, n := range s.Notes { + if n.Session == id { + out = append(out, Turn{Role: "vr", Kind: "note", When: n.At, File: n.File, Note: n}) + } + } + return out +} + +// conversation merges a session's transcript with its log entries. Both sides +// are compared as instants — the log stamps carry an offset, transcripts are +// UTC — so ordering is the real one, and only the rendering is local. +func (sv *server) conversation(site *Site, id string, note int, all bool) *Conv { + c := &Conv{Session: id, Roots: sv.tx.roots} + if err := safeSession(id); err != nil { + c.Err = err.Error() + return c + } + for _, s := range site.Sessions(time.Now(), liveWindow, 0) { + if s.ID == id { + c.Agent, c.Model, c.Dir, c.Bit = s.Agent, s.Model, s.Dir, s.Bit + c.Reads, c.Greps, c.Notes = s.Reads, s.Greps, s.Notes + break + } + } + + logged := site.sessionTurns(id) + // the parsed transcript is shared with other requests, so the merge copies + // it rather than appending onto its slice + var turns []Turn + if t := sv.tx.load(id, c.Dir); t != nil { + c.Path, c.Harness, c.Err = t.Path, t.Harness, t.Err + turns = make([]Turn, 0, len(t.Turns)+len(logged)) + turns = append(turns, t.Turns...) + } + turns = append(turns, logged...) + sort.SliceStable(turns, func(i, j int) bool { return turns[i].When.Before(turns[j].When) }) + + at := -1 + if note > 0 { + for i := range turns { + if turns[i].Note != nil && turns[i].Note.ID == note { + at, turns[i].Here = i, true + c.Anchor = fmt.Sprintf("note-%d", note) + break + } + } + } + c.Total = len(turns) + from, window := convWindowAt(turns, at, all) + c.Turns, c.From, c.To = window, from+1, from+len(window) + c.Windowed = len(window) < len(turns) + c.Incomplete = c.Windowed && all + c.AllHref = "/session/" + escPath(id) + "?all=1" + if note > 0 { + c.AllHref += fmt.Sprintf("¬e=%d#note-%d", note, note) + } + return c +} + +// convWindowAt cuts a conversation down to what a page may render: a stretch +// around the anchored moment, or the tail of it when nothing is anchored. +// "show all" raises the bound but does not remove it. +func convWindowAt(turns []Turn, at int, all bool) (int, []Turn) { + n := convWindow + if all { + n = convMax + } + if len(turns) <= n { + return 0, turns + } + if at < 0 { + return len(turns) - n, turns[len(turns)-n:] + } + lo := at - n/2 + if lo < 0 { + lo = 0 + } + if lo+n > len(turns) { + lo = len(turns) - n + } + return lo, turns[lo : lo+n] +} |
