summaryrefslogtreecommitdiff
path: root/web/render.go
diff options
context:
space:
mode:
Diffstat (limited to 'web/render.go')
-rw-r--r--web/render.go462
1 files changed, 462 insertions, 0 deletions
diff --git a/web/render.go b/web/render.go
new file mode 100644
index 0000000..0bfbdce
--- /dev/null
+++ b/web/render.go
@@ -0,0 +1,462 @@
+// render.go — source to HTML: tree-sitter highlighting, coverage classes and
+// note anchors baked into one <ol> per chunk.
+package web
+
+import (
+ "bytes"
+ "html/template"
+ "path/filepath"
+ "sort"
+ "strconv"
+ "strings"
+ "unsafe"
+
+ tszig "github.com/tree-sitter-grammars/tree-sitter-zig/bindings/go"
+ sitter "github.com/tree-sitter/go-tree-sitter"
+ tsbash "github.com/tree-sitter/tree-sitter-bash/bindings/go"
+ tsc "github.com/tree-sitter/tree-sitter-c/bindings/go"
+ tscpp "github.com/tree-sitter/tree-sitter-cpp/bindings/go"
+ tsgo "github.com/tree-sitter/tree-sitter-go/bindings/go"
+ tsjs "github.com/tree-sitter/tree-sitter-javascript/bindings/go"
+ tsjson "github.com/tree-sitter/tree-sitter-json/bindings/go"
+ tspython "github.com/tree-sitter/tree-sitter-python/bindings/go"
+ tsrust "github.com/tree-sitter/tree-sitter-rust/bindings/go"
+ tsts "github.com/tree-sitter/tree-sitter-typescript/bindings/go"
+)
+
+// Lines per content-visibility chunk: the browser skips layout and paint of
+// offscreen chunks wholesale, instead of managing one containment context
+// per line.
+const chunkLines = 800
+
+// approximate rendered line height, for contain-intrinsic-size placeholders
+const lineHeightPx = 19
+
+// render builds the code HTML: <ol> chunks of chunkLines lines, each line an
+// <li> with syntax spans, coverage classes and note anchors. One pre-sized
+// builder, no fmt, minimal escaping — large files are browser-bound, so the
+// output is kept as small and flat as possible.
+func (f *File) render(src []byte, hl *highlighter) {
+ classes := hl.classify(f.Path, src)
+ f.Total = bytes.Count(src, []byte{'\n'})
+ if len(src) > 0 && src[len(src)-1] != '\n' {
+ f.Total++
+ }
+ f.Covered = 0
+ for i := 1; i < len(f.Cov) && i <= f.Total; i++ {
+ if f.Cov[i] != 0 {
+ f.Covered++
+ }
+ }
+ if f.Total > 0 {
+ f.Pct = f.Covered * 100 / f.Total
+ }
+ sort.Slice(f.Notes, func(i, j int) bool { return f.Notes[i].Start < f.Notes[j].Start })
+ for ni, n := range f.Notes {
+ for l := n.Start; l <= n.End && l > 0; l++ {
+ if _, taken := f.NoteAt[l]; !taken {
+ f.NoteAt[l] = ni
+ }
+ }
+ }
+ f.Body = f.emit(src, classes, hl.classNames, 1, f.Total)
+}
+
+// emit writes lines [from..to] as <ol> chunks. from is where the chunking (and
+// so the browser's content-visibility windows) starts counting, which is what
+// lets an excerpt come out as one chunk rather than a slice of the file's.
+func (f *File) emit(src []byte, classes []int16, names []string, from, to int) template.HTML {
+ if from < 1 {
+ from = 1
+ }
+ if to > f.Total {
+ to = f.Total
+ }
+ if from > to {
+ return ""
+ }
+ start := 0
+ for ln := 1; ln < from; ln++ { // skip to the first line asked for
+ start = pastEOL(src, start)
+ }
+ stop := start
+ for ln := from; ln <= to; ln++ {
+ stop = pastEOL(src, stop)
+ }
+ var b strings.Builder
+ span := stop - start
+ b.Grow(span + span/2 + (to-from+1)*48)
+ var num []byte
+ writeInt := func(n int) { num = strconv.AppendInt(num[:0], int64(n), 10); b.Write(num) }
+ for ln := from; ln <= to; ln++ {
+ end := start
+ for end < len(src) && src[end] != '\n' {
+ end++
+ }
+ opens := (ln-from)%chunkLines == 0
+ if opens {
+ if ln > from {
+ b.WriteString("</ol>")
+ }
+ rem := to - (ln - 1)
+ if rem > chunkLines {
+ rem = chunkLines
+ }
+ b.WriteString(`<ol class="code" style="--h: `)
+ writeInt(rem * lineHeightPx)
+ b.WriteString(`px">`)
+ }
+ b.WriteString(`<li id="L`)
+ writeInt(ln)
+ b.WriteByte('"')
+ // Where the chunk's numbering starts. It has to be counter-set on the
+ // first line rather than counter-reset on the <ol>, because
+ // content-visibility: auto brings style containment with it and a
+ // contained element's own counter-reset comes out as 0 no matter what
+ // value it names — so every chunk but the first would count from 1.
+ if opens {
+ b.WriteString(` style="counter-set: ln `)
+ writeInt(ln)
+ b.WriteByte('"')
+ }
+ mask := uint8(0)
+ if ln < len(f.Cov) {
+ mask = f.Cov[ln]
+ }
+ ni, noted := f.NoteAt[ln]
+ if mask != 0 || noted {
+ b.WriteString(` class="`)
+ if mask != 0 {
+ b.WriteString("cov am")
+ writeInt(int(mask))
+ if noted {
+ b.WriteString(" noted")
+ }
+ } else {
+ b.WriteString("noted")
+ }
+ b.WriteByte('"')
+ }
+ if noted {
+ b.WriteString(` data-note="note-`)
+ writeInt(f.Notes[ni].ID)
+ b.WriteByte('"')
+ }
+ b.WriteByte('>')
+ emitLine(&b, src[start:end], classes[start:end], names)
+ b.WriteString("</li>")
+ start = end + 1
+ }
+ b.WriteString("</ol>")
+ return template.HTML(b.String())
+}
+
+// pastEOL is the offset just past the line starting at i.
+func pastEOL(src []byte, i int) int {
+ for i < len(src) && src[i] != '\n' {
+ i++
+ }
+ if i < len(src) {
+ i++
+ }
+ return i
+}
+
+// excerptContext is how many lines either side of a note's range its own page
+// shows, so the noted lines are read in context rather than in isolation.
+const excerptContext = 4
+
+// excerptFor renders just the lines a note is about, the way the code view
+// would: the file's real line numbers, its coverage tint, and the noted lines
+// marked. This is what lets a shared permalink stand on its own — whoever
+// opens it sees the note and the code it is about together, without having to
+// find their way into the file first.
+//
+// cov comes from the model and is read-only here, so the caller can pass the
+// live *File's slice rather than copying it.
+func excerptFor(path string, src []byte, cov []uint8, hl *highlighter, n *Note) (body template.HTML, from, to int) {
+ total := bytes.Count(src, []byte{'\n'})
+ if len(src) > 0 && src[len(src)-1] != '\n' {
+ total++
+ }
+ end := n.End
+ if end < n.Start {
+ end = n.Start
+ }
+ if n.Start < 1 || total == 0 {
+ return "", 0, 0
+ }
+ f := &File{Path: path, Cov: cov, Total: total, Notes: []*Note{n}, NoteAt: map[int]int{}}
+ for l := n.Start; l <= end; l++ {
+ f.NoteAt[l] = 0
+ }
+ from, to = n.Start-excerptContext, end+excerptContext
+ if from < 1 {
+ from = 1
+ }
+ if to > total {
+ to = total
+ }
+ if from > total {
+ return "", 0, 0
+ }
+ return f.emit(src, hl.classify(path, src), hl.classNames, from, to), from, to
+}
+
+// emitLine writes one line's spans, merging runs of the same class across
+// whitespace-only gaps ("pub fn" is one span, not two) to keep the DOM flat.
+func emitLine(b *strings.Builder, line []byte, cls []int16, names []string) {
+ open := int16(0)
+ i := 0
+ for i < len(line) {
+ j := i
+ c := cls[i]
+ for j < len(line) && cls[j] == c {
+ j++
+ }
+ if c == 0 {
+ allSpace := true
+ for k := i; k < j; k++ {
+ if line[k] != ' ' && line[k] != '\t' {
+ allSpace = false
+ break
+ }
+ }
+ if !(allSpace && open != 0 && j < len(line) && cls[j] == open) {
+ if open != 0 {
+ b.WriteString("</span>")
+ open = 0
+ }
+ }
+ escapeTo(b, line[i:j])
+ } else {
+ if open != c {
+ if open != 0 {
+ b.WriteString("</span>")
+ }
+ b.WriteString(`<span class="s-`)
+ b.WriteString(names[c])
+ b.WriteString(`">`)
+ open = c
+ }
+ escapeTo(b, line[i:j])
+ }
+ i = j
+ }
+ if open != 0 {
+ b.WriteString("</span>")
+ }
+}
+
+// escapeTo escapes the three characters that matter in text content.
+func escapeTo(b *strings.Builder, s []byte) {
+ last := 0
+ for i, c := range s {
+ var rep string
+ switch c {
+ case '&':
+ rep = "&amp;"
+ case '<':
+ rep = "&lt;"
+ case '>':
+ rep = "&gt;"
+ default:
+ continue
+ }
+ b.Write(s[last:i])
+ b.WriteString(rep)
+ last = i + 1
+ }
+ b.Write(s[last:])
+}
+
+// ── highlighting ──────────────────────────────────────────────────────────
+
+// grammars lists every language the site can highlight: the extensions it
+// claims, the vendored queries that paint it, and the cgo grammar. Query files
+// concatenate the way upstream composes them — C's patterns underpin C++,
+// JavaScript's underpin TypeScript — and the concatenation keeps the
+// later-patterns-win order, so the supplement wins where both match.
+var grammars = []struct {
+ exts []string
+ scms []string
+ lang func() unsafe.Pointer
+}{
+ {[]string{".rs"}, []string{"rust"}, tsrust.Language},
+ {[]string{".go"}, []string{"go"}, tsgo.Language},
+ {[]string{".py", ".pyi"}, []string{"python"}, tspython.Language},
+ {[]string{".js", ".mjs", ".cjs", ".jsx"}, []string{"javascript", "javascript-params", "jsx"}, tsjs.Language},
+ {[]string{".ts", ".mts", ".cts"}, []string{"javascript", "typescript"}, tsts.LanguageTypescript},
+ {[]string{".tsx"}, []string{"javascript", "typescript", "jsx"}, tsts.LanguageTSX},
+ {[]string{".c"}, []string{"c"}, tsc.Language},
+ // .h is C++'s upstream claim, and the C++ grammar is a superset: C headers
+ // parse the same under it, C++ headers only parse under it
+ {[]string{".cc", ".cpp", ".cxx", ".h", ".hh", ".hpp", ".hxx"}, []string{"c", "cpp"}, tscpp.Language},
+ {[]string{".zig"}, []string{"zig"}, tszig.Language},
+ {[]string{".json"}, []string{"json"}, tsjson.Language},
+ {[]string{".sh", ".bash"}, []string{"bash"}, tsbash.Language},
+}
+
+// language is one grammar's share of the highlighter. Everything here is built
+// at startup and never written again, which is what lets classify run from any
+// number of goroutines at once.
+type language struct {
+ lang *sitter.Language
+ query *sitter.Query
+ captureCls []int16 // capture index -> class id
+}
+
+type highlighter struct {
+ byExt map[string]*language
+ classNames []string
+}
+
+// resolveKey finds the theme syntax key for a capture name by stripping dot
+// segments; "" when the theme has no entry at all.
+func resolveKey(t *Theme, name string) string {
+ for key := name; key != ""; {
+ if s, ok := t.Syntax[key]; ok && s.Color != "" {
+ return key
+ }
+ if i := strings.LastIndex(key, "."); i >= 0 {
+ key = key[:i]
+ } else {
+ break
+ }
+ }
+ return ""
+}
+
+// nearFg reports whether a capture would render indistinguishably from plain
+// foreground text in the theme. Spans for such captures are pure DOM weight.
+func nearFg(t *Theme, name string) bool {
+ key := resolveKey(t, name)
+ if key == "" {
+ return true // no color -> falls through to fg anyway
+ }
+ return colorDist(hex6(t.Syntax[key].Color), hex6(t.Fg)) < 2000
+}
+
+func newHighlighter(dark, light *Theme) *highlighter {
+ h := &highlighter{byExt: map[string]*language{}, classNames: []string{""}}
+ for _, g := range grammars {
+ var scm strings.Builder
+ for _, name := range g.scms {
+ b, err := tfs.ReadFile("assets/" + name + ".scm")
+ if err != nil {
+ fatal("%v", err)
+ }
+ scm.Write(b)
+ }
+ l := &language{lang: sitter.NewLanguage(g.lang())}
+ q, qerr := sitter.NewQuery(l.lang, scm.String())
+ if qerr != nil {
+ fatal("%s query: %v", strings.Join(g.scms, "+"), qerr)
+ }
+ l.query = q
+ for _, name := range q.CaptureNames() {
+ l.captureCls = append(l.captureCls, h.classFor(dark, light, name))
+ }
+ for _, ext := range g.exts {
+ h.byExt[ext] = l
+ }
+ }
+ return h
+}
+
+// classFor interns the class id of a capture name, which doubles as a zed
+// theme syntax key. Captures both themes paint (nearly) as foreground get id 0
+// and no span at all — punctuation and friends are the bulk of all spans.
+func (h *highlighter) classFor(dark, light *Theme, name string) int16 {
+ if nearFg(dark, name) && nearFg(light, name) {
+ return 0
+ }
+ key := resolveKey(dark, name)
+ if key == "" {
+ key = resolveKey(light, name)
+ }
+ cls := strings.ReplaceAll(key, ".", "-")
+ for i, n := range h.classNames {
+ if n == cls {
+ return int16(i)
+ }
+ }
+ h.classNames = append(h.classNames, cls)
+ return int16(len(h.classNames) - 1)
+}
+
+type paint struct {
+ start, end uint
+ pattern uint
+ cls int16
+}
+
+// classify paints a file. Extensions no grammar claims come back unclassified
+// rather than guessed at.
+func (h *highlighter) classify(path string, src []byte) []int16 {
+ return h.classifyWith(h.byExt[strings.ToLower(filepath.Ext(path))], src)
+}
+
+// fenceAliases maps the fence languages that are not already one of the
+// extensions in grammars. Most are ("go", "rs", "py", "ts", "json", "bash"),
+// so the grammar table does the bulk of the lookup on its own.
+var fenceAliases = map[string]string{
+ "golang": ".go", "rust": ".rs", "python": ".py", "python3": ".py",
+ "javascript": ".js", "node": ".js", "typescript": ".ts",
+ "c++": ".cc", "shell": ".sh", "zsh": ".sh", "console": ".sh",
+ "jsonc": ".json", "json5": ".json",
+}
+
+// classifyLang paints a fenced code block, whose language arrives as a name
+// rather than a path. Unknown or absent languages come back unclassified, the
+// same as an extension no grammar claims.
+func (h *highlighter) classifyLang(lang string, src []byte) []int16 {
+ if h == nil {
+ return make([]int16, len(src))
+ }
+ name := strings.ToLower(strings.TrimSpace(lang))
+ l := h.byExt["."+name]
+ if l == nil {
+ l = h.byExt[fenceAliases[name]]
+ }
+ return h.classifyWith(l, src)
+}
+
+// classifyWith is safe for concurrent use: the shared per-language query is
+// immutable, and parser and cursor are per-call.
+//
+// Matches arrive already filtered by their #eq?/#match?/#any-of? predicates —
+// go-tree-sitter evaluates those in QueryMatches.Next — so every capture here
+// is one the query really meant.
+func (h *highlighter) classifyWith(l *language, src []byte) []int16 {
+ classes := make([]int16, len(src))
+ if l == nil {
+ return classes
+ }
+ parser := sitter.NewParser()
+ defer parser.Close()
+ parser.SetLanguage(l.lang)
+ tree := parser.Parse(src, nil)
+ defer tree.Close()
+
+ qc := sitter.NewQueryCursor()
+ defer qc.Close()
+ matches := qc.Matches(l.query, tree.RootNode(), src)
+ var paints []paint
+ for m := matches.Next(); m != nil; m = matches.Next() {
+ for _, c := range m.Captures {
+ if cls := l.captureCls[c.Index]; cls != 0 {
+ paints = append(paints, paint{c.Node.StartByte(), c.Node.EndByte(), m.PatternIndex, cls})
+ }
+ }
+ }
+ // later patterns in the query paint over earlier ones; two captures of one
+ // pattern must not overlap, since this sort leaves their order arbitrary
+ sort.Slice(paints, func(i, j int) bool { return paints[i].pattern < paints[j].pattern })
+ for _, p := range paints {
+ for i := p.start; i < p.end && int(i) < len(classes); i++ {
+ classes[i] = p.cls
+ }
+ }
+ return classes
+}