summaryrefslogtreecommitdiff
path: root/web/bench_test.go
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-02 23:43:08 -0300
committerGabriel Schneider <[email protected]>2026-08-03 09:54:39 -0300
commit5ec72f8723d795d0b02d7e9ebb9f555f0c7e6a2e (patch)
treef12f4b700fff4ff6f7e818d2a542672ba0b8d76e /web/bench_test.go
parenta9263daee9413c5eff3c1f1bebcd224442fd8685 (diff)
downloadnotevi-5ec72f8723d795d0b02d7e9ebb9f555f0c7e6a2e.tar.gz
notevi-5ec72f8723d795d0b02d7e9ebb9f555f0c7e6a2e.zip
rebrand to notevi: one CLI over the jj sidecar
vr and vrsite become a single binary. vrsite/ folds into a web package in one module (0x4200.cafe/notevi); "notevi web" serves and exports exactly what vrsite did, and "notevi read/grep/note/query" is unchanged. The sidecar is renamed with it: notevi_log, notevi-log.jsonl, and the description "private: notevi log". The pre-rebrand names are still recognized, so an old repository opens and reads; it is renamed in place on the first write, or up front with "notevi migrate DIR...". That rename cannot be a single mv inside jj run. jj only auto-tracks a *new* file in the run working copy below a size limit it does not take from the command line, so writing a whole log under a name the change has never held is silently dropped while jj reports success. ensureLogFile creates the file empty first and lets every later byte be a modification of a tracked file, which snapshots at any size; that also fixes the same latent bug when importing a large legacy vr-log.jsonl. Adds a bem-te-vi mark (favicon and nav brand) and a README. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
Diffstat (limited to 'web/bench_test.go')
-rw-r--r--web/bench_test.go328
1 files changed, 328 insertions, 0 deletions
diff --git a/web/bench_test.go b/web/bench_test.go
new file mode 100644
index 0000000..094eb47
--- /dev/null
+++ b/web/bench_test.go
@@ -0,0 +1,328 @@
+package web
+
+import (
+ "bytes"
+ "fmt"
+ "io"
+ "testing"
+)
+
+// A realistic Rust chunk (~1.2KB): doc comments, generics, lifetimes, attrs,
+// strings with escapes, macros — the constructs the highlighter works hardest
+// on. Repetition parses fine; tree-sitter is not a compiler.
+var rustChunk = []byte(`//! Module documentation with some text.
+use std::{collections::HashMap, sync::Arc};
+
+/// A cached entry with generics and lifetimes.
+#[derive(Debug, Clone, PartialEq)]
+pub struct CacheEntry<'a, T: Clone + Send> {
+ pub key: &'a str,
+ pub value: Arc<T>,
+ hits: u64,
+ tags: HashMap<String, Vec<u8>>,
+}
+
+pub enum LookupResult<T> {
+ Hit(Arc<T>),
+ Miss { reason: &'static str, code: i32 },
+ Tombstone,
+}
+
+impl<'a, T: Clone + Send> CacheEntry<'a, T> {
+ pub fn new(key: &'a str, value: T) -> Self {
+ let tags = HashMap::new();
+ Self { key, value: Arc::new(value), hits: 0, tags }
+ }
+
+ /// Record a hit and return the running total.
+ pub fn touch(&mut self) -> u64 {
+ self.hits = self.hits.saturating_add(1);
+ if self.hits % 100 == 0 {
+ println!("entry {} hit {} times \"escaped\"", self.key, self.hits);
+ }
+ self.hits
+ }
+
+ pub fn lookup(&self, keys: &[&str]) -> LookupResult<T> {
+ match keys.iter().position(|k| *k == self.key) {
+ Some(_) => LookupResult::Hit(self.value.clone()),
+ None => LookupResult::Miss { reason: "absent <key>", code: -1 },
+ }
+ }
+}
+
+const MAX_ENTRIES: usize = 4096;
+static GREETING: &str = "hello \"quoted\" <world> & friends";
+`)
+
+// One chunk per language whose query leans on a predicate, to keep an eye on
+// what predicates cost: go's builtin list is an #any-of? (a byte compare per
+// call), while python, javascript and C run a casing regex over every
+// identifier they see — and zig, which names its types by casing alone, runs
+// two.
+var langChunks = []struct {
+ path string
+ src []byte
+}{
+ {"bench.go", []byte(`// Package cache keeps entries with hit counts.
+package cache
+
+import (
+ "fmt"
+ "sync"
+)
+
+// Entry is one cached value and its hit count.
+type Entry struct {
+ Key string
+ Value []byte
+ hits uint64
+ tags map[string][]byte
+}
+
+const MaxEntries = 4096
+
+func NewEntry(key string, value []byte) *Entry {
+ return &Entry{Key: key, Value: value, tags: make(map[string][]byte)}
+}
+
+// Touch records a hit and returns the running total.
+func (e *Entry) Touch(mu *sync.Mutex) uint64 {
+ mu.Lock()
+ defer mu.Unlock()
+ e.hits++
+ if e.hits%100 == 0 {
+ fmt.Printf("entry %q hit %d times \"escaped\"\n", e.Key, e.hits)
+ }
+ return e.hits
+}
+
+func (e *Entry) Lookup(keys []string) ([]byte, bool) {
+ for _, k := range keys {
+ if v, ok := e.tags[k]; ok && len(v) > 0 {
+ return v, true
+ }
+ }
+ return nil, false
+}
+`)},
+ {"bench.py", []byte(`"""Module documentation with some text."""
+import sys
+from collections import OrderedDict
+
+MAX_ENTRIES = 4096
+
+
+class CacheEntry:
+ """A cached entry with a hit count."""
+
+ def __init__(self, key, value):
+ self.key = key
+ self.value = value
+ self.hits = 0
+ self.tags = OrderedDict()
+
+ @property
+ def stale(self):
+ return self.hits > MAX_ENTRIES
+
+ def touch(self):
+ self.hits += 1
+ if self.hits % 100 == 0:
+ print(f"entry {self.key} hit {self.hits} times \"escaped\"", file=sys.stderr)
+ return self.hits
+
+ def lookup(self, keys):
+ for k in keys:
+ if k in self.tags:
+ return self.tags[k], True
+ return None, False
+`)},
+ {"bench.js", []byte(`// Cache entries with hit counts.
+import { EventEmitter } from "node:events";
+
+const MAX_ENTRIES = 4096;
+
+export class CacheEntry extends EventEmitter {
+ constructor(key, value) {
+ super();
+ this.key = key;
+ this.value = value;
+ this.hits = 0;
+ this.tags = new Map();
+ }
+
+ get stale() {
+ return this.hits > MAX_ENTRIES;
+ }
+
+ touch() {
+ this.hits += 1;
+ if (this.hits % 100 === 0) {
+ console.log(` + "`entry ${this.key} hit ${this.hits} times \"escaped\"`" + `);
+ }
+ return this.hits;
+ }
+
+ lookup(keys) {
+ return keys.map((k) => this.tags.get(k)).find((v) => v !== undefined) ?? null;
+ }
+}
+`)},
+ {"bench.c", []byte(`/* Cache entries with hit counts. */
+#include <stdio.h>
+#include <string.h>
+
+#define MAX_ENTRIES 4096
+
+struct cache_entry {
+ const char *key;
+ unsigned char *value;
+ unsigned long hits;
+};
+
+static struct cache_entry *entry_new(const char *key, unsigned char *value) {
+ static struct cache_entry e;
+ e.key = key;
+ e.value = value;
+ e.hits = 0;
+ return &e;
+}
+
+unsigned long entry_touch(struct cache_entry *e) {
+ e->hits++;
+ if (e->hits % 100 == 0) {
+ fprintf(stderr, "entry %s hit %lu times \"escaped\"\n", e->key, e->hits);
+ }
+ return e->hits;
+}
+
+int entry_lookup(struct cache_entry *e, const char *key) {
+ return e->key != NULL && strcmp(e->key, key) == 0;
+}
+`)},
+ {"bench.zig", []byte(`// Cache entries with hit counts.
+const std = @import("std");
+
+const MAX_ENTRIES: usize = 4096;
+
+const CacheEntry = struct {
+ key: []const u8,
+ value: []u8,
+ hits: u64 = 0,
+
+ fn init(key: []const u8, value: []u8) CacheEntry {
+ return CacheEntry{ .key = key, .value = value };
+ }
+
+ fn stale(self: CacheEntry) bool {
+ return self.hits > MAX_ENTRIES;
+ }
+
+ fn touch(self: *CacheEntry) u64 {
+ self.hits += 1;
+ if (self.hits % 100 == 0) {
+ std.debug.print("entry {s} hit {d} times \"escaped\"\n", .{ self.key, self.hits });
+ }
+ return self.hits;
+ }
+
+ fn lookup(self: CacheEntry, keys: []const []const u8) ?[]u8 {
+ for (keys) |k| {
+ if (std.mem.eql(u8, k, self.key)) return self.value;
+ }
+ return null;
+ }
+};
+`)},
+}
+
+func synth(chunk []byte, size int) []byte {
+ var b bytes.Buffer
+ for b.Len() < size {
+ b.Write(chunk)
+ }
+ return b.Bytes()
+}
+
+func synthRust(size int) []byte { return synth(rustChunk, size) }
+
+var sizes = []int{128 << 10, 1 << 20, 4 << 20}
+
+func benchHL(b *testing.B) *highlighter {
+ b.Helper()
+ return newHighlighter(loadTheme("", "", "dark"), loadTheme("", "", "light"))
+}
+
+// tree-sitter parse + query + paint only
+func BenchmarkClassify(b *testing.B) {
+ hl := benchHL(b)
+ for _, size := range sizes {
+ src := synthRust(size)
+ b.Run(fmt.Sprintf("%dKB", size>>10), func(b *testing.B) {
+ b.SetBytes(int64(len(src)))
+ b.ReportAllocs()
+ for i := 0; i < b.N; i++ {
+ hl.classify("bench.rs", src)
+ }
+ })
+ }
+}
+
+// the same work for the languages whose queries evaluate predicates
+func BenchmarkClassifyLangs(b *testing.B) {
+ hl := benchHL(b)
+ for _, c := range langChunks {
+ src := synth(c.src, 1<<20)
+ b.Run(c.path, func(b *testing.B) {
+ b.SetBytes(int64(len(src)))
+ b.ReportAllocs()
+ for i := 0; i < b.N; i++ {
+ hl.classify(c.path, src)
+ }
+ })
+ }
+}
+
+// classify + full per-line HTML emit
+func BenchmarkRender(b *testing.B) {
+ hl := benchHL(b)
+ for _, size := range sizes {
+ src := synthRust(size)
+ b.Run(fmt.Sprintf("%dKB", size>>10), func(b *testing.B) {
+ b.SetBytes(int64(len(src)))
+ b.ReportAllocs()
+ for i := 0; i < b.N; i++ {
+ f := &File{Path: "bench.rs", NoteAt: map[int]int{}}
+ f.render(src, hl)
+ }
+ })
+ }
+}
+
+// template execution for an already-rendered file page
+func BenchmarkExecute(b *testing.B) {
+ hl := benchHL(b)
+ tpl := newTemplates(hl)
+ site := &Site{
+ Title: "bench", Files: map[string]*File{},
+ Theme: loadTheme("", "", "dark"), Light: loadTheme("", "", "light"),
+ Agents: []*Agent{{Name: "bench", Short: "bench", Bit: 1}},
+ }
+ for _, size := range sizes {
+ src := synthRust(size)
+ f := &File{Path: "bench.rs", NoteAt: map[int]int{}, Traced: true, Change: "zzzzzzzz"}
+ f.cover(1, bytes.Count(src, []byte("\n"))/2, 1)
+ f.render(src, hl)
+ p := &page{Site: site, Kind: "file", Title: "bench", Root: "", Current: f.Path, File: f}
+ b.Run(fmt.Sprintf("%dKB", size>>10), func(b *testing.B) {
+ b.SetBytes(int64(len(src)))
+ b.ReportAllocs()
+ for i := 0; i < b.N; i++ {
+ if err := tpl.Execute(io.Discard, p); err != nil {
+ b.Fatal(err)
+ }
+ }
+ })
+ }
+}