//! Tree-sitter syntax highlighting: one style byte per content byte, filled by //! running each grammar's highlights.scm query (slurped at build time into the //! ts_queries options module). Grammar set is tiered: `zig`; `minimal` (c, cpp, //! zig); and `full`, which adds ~24 languages lazily on first use. const std = @import("std"); const config = @import("pardes_config"); const tracy = @import("tracy.zig"); const grammar_manifest = @import("grammar_manifest.zig"); const look = @import("look.zig"); pub const enabled = config.syntax_highlighting; const zig_grammar = config.syntax_zig_grammar; const minimal_grammars = config.syntax_minimal_grammars; const full_grammars = config.syntax_full_grammars; const ts = if (enabled) @import("tree-sitter") else struct { pub const Language = opaque {}; pub const Query = opaque {}; }; const ts_queries = if (enabled) @import("ts_queries") else struct {}; const LanguageFn = *const fn () callconv(.c) *const ts.Language; pub const Syn = enum(u8) { none, keyword, string, number, comment }; const Spec = struct { name: []const u8, exts: []const []const u8, language: *const fn () callconv(.c) *const ts.Language, query_src: []const u8, compiled_query: ?*ts.Query = null, }; fn grammarSelected(comptime g: grammar_manifest.Grammar) bool { return switch (g.tier) { .zig => zig_grammar, .minimal => minimal_grammars, .full => full_grammars, }; } fn specCount() comptime_int { var count = 0; for (grammar_manifest.all) |g| { if (grammarSelected(g)) count += 1; } return count; } // Upstream's typst highlights.scm names its markup honestly — // @markup.heading.*, @markup.bold, @markup.italic, @markup.raw.block — and // `synFor` now understands that vocabulary, so headings, bold, italics and raw // blocks paint on their own. What is left here is exactly the two captures we // refuse to map globally: // // - upstream tags call callees @function/@function.method; mapping "function" // in `synFor` would recolour every function call in every language. // - upstream tags the code sigil "#" @operator, and mapping "operator" would // likewise light up every +, -, == in the codebase. // // Both are worth colouring *in typst specifically*: the sigil in front of every // #let/#if/#import/#call is the visual anchor of the code/markup split, and // upstream leaves it uncoloured even though the keyword behind it is not. const typst_supplement = \\ \\(call item: (ident) @keyword) \\(call item: (field field: (ident) @keyword)) \\"#" @keyword \\ ; fn querySrc(comptime g: grammar_manifest.Grammar) []const u8 { const base = @field(ts_queries, g.name ++ "_highlights"); if (comptime std.mem.eql(u8, g.name, "typst")) return base ++ typst_supplement; return base; } fn initSpecs() [specCount()]Spec { var out: [specCount()]Spec = undefined; var i = 0; inline for (grammar_manifest.all) |g| { if (grammarSelected(g)) { out[i] = .{ .name = g.name, .exts = g.exts, .language = @extern(LanguageFn, .{ .name = "tree_sitter_" ++ g.name }), .query_src = querySrc(g), }; i += 1; } } return out; } var specs = initSpecs(); const allocation_header_size = 16; var syntax_allocator: std.mem.Allocator = undefined; var syntax_started = false; extern fn ts_set_allocator( new_malloc: ?*const fn (size: usize) callconv(.c) ?*anyopaque, new_calloc: ?*const fn (nmemb: usize, size: usize) callconv(.c) ?*anyopaque, new_realloc: ?*const fn (ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque, new_free: ?*const fn (ptr: ?*anyopaque) callconv(.c) void, ) void; fn syntaxAlloc(size_arg: usize) callconv(.c) ?*anyopaque { const size = @max(size_arg, 1); const total = std.math.add(usize, allocation_header_size, size) catch return null; const bytes = syntax_allocator.alignedAlloc(u8, .@"16", total) catch return null; const header: *align(16) usize = @ptrCast(bytes.ptr); header.* = total; return @ptrCast(bytes.ptr + allocation_header_size); } fn syntaxCalloc(count: usize, size: usize) callconv(.c) ?*anyopaque { const len = std.math.mul(usize, count, size) catch return null; const pointer = syntaxAlloc(len) orelse return null; const bytes: [*]u8 = @ptrCast(pointer); @memset(bytes[0..len], 0); return pointer; } fn syntaxRealloc(ptr: ?*anyopaque, new_size: usize) callconv(.c) ?*anyopaque { const pointer = ptr orelse return syntaxAlloc(new_size); if (new_size == 0) { syntaxFree(pointer); return null; } const user: [*]u8 = @ptrCast(pointer); const header: *align(16) usize = @ptrCast(@alignCast(user - allocation_header_size)); const old_total = header.*; const old_bytes: []align(16) u8 = @as([*]align(16) u8, @ptrCast(header))[0..old_total]; const new_total = std.math.add(usize, allocation_header_size, new_size) catch return null; const new_bytes = syntax_allocator.realloc(old_bytes, new_total) catch return null; const new_header: *align(16) usize = @ptrCast(new_bytes.ptr); new_header.* = new_total; return @ptrCast(new_bytes.ptr + allocation_header_size); } fn syntaxFree(ptr: ?*anyopaque) callconv(.c) void { const pointer = ptr orelse return; const user: [*]u8 = @ptrCast(pointer); const header: *align(16) usize = @ptrCast(@alignCast(user - allocation_header_size)); const total = header.*; const bytes: []align(16) u8 = @as([*]align(16) u8, @ptrCast(header))[0..total]; syntax_allocator.free(bytes); } pub fn start(gpa: std.mem.Allocator) void { if (comptime enabled) { std.debug.assert(!syntax_started); syntax_allocator = gpa; syntax_started = true; ts_set_allocator(syntaxAlloc, syntaxCalloc, syntaxRealloc, syntaxFree); } } pub fn stop() void { if (comptime enabled) { std.debug.assert(syntax_started); for (&specs) |*spec| { if (spec.compiled_query) |query| query.destroy(); spec.compiled_query = null; } ts_set_allocator(null, null, null, null); syntax_started = false; syntax_allocator = undefined; } } const Selected = struct { name: []const u8, lang: *const ts.Language, query: *ts.Query }; // NOTE: don't lang.destroy() — the tree_sitter_*() languages are static // singletons reused on every open; destroying one use-after-frees the next. fn ensure(spec: *Spec) !Selected { const lang = spec.language(); if (spec.compiled_query) |query| return .{ .name = spec.name, .lang = lang, .query = query }; var error_offset: u32 = 0; const query = try ts.Query.create(lang, spec.query_src, &error_offset); spec.compiled_query = query; return .{ .name = spec.name, .lang = lang, .query = query }; } fn forExt(ext: []const u8) !?Selected { for (&specs) |*spec| { for (spec.exts) |choice| { if (std.ascii.eqlIgnoreCase(ext, choice)) return try ensure(spec); } } return null; } const LangAlias = struct { []const u8, []const u8 }; const lang_aliases = [_]LangAlias{ .{ "js", "javascript" }, .{ "jsx", "javascript" }, .{ "mjs", "javascript" }, .{ "py", "python" }, .{ "sh", "bash" }, .{ "shell", "bash" }, .{ "zsh", "bash" }, .{ "bash", "bash" }, .{ "rs", "rust" }, .{ "c++", "cpp" }, .{ "cxx", "cpp" }, .{ "cc", "cpp" }, .{ "cs", "c_sharp" }, .{ "kt", "kotlin" }, .{ "rb", "ruby" }, .{ "ml", "ocaml" }, .{ "hs", "haskell" }, }; fn forLang(name: []const u8) !?Selected { var canonical = name; for (lang_aliases) |a| { if (std.ascii.eqlIgnoreCase(name, a[0])) { canonical = a[1]; break; } } for (&specs) |*spec| { if (std.ascii.eqlIgnoreCase(canonical, spec.name)) return try ensure(spec); } return null; } /// The caller owns the cursor: `ts_query_cursor_exec` fully resets its state, /// so one cursor serves any number of trees, and the per-row pass would /// otherwise create and destroy one — three allocations against the shared /// tree-sitter arena — for every row of a results buffer. fn runQuery(styles: []u8, sel: Selected, tree: *ts.Tree, base: usize, cursor: *ts.QueryCursor) void { cursor.exec(sel.query, tree.rootNode()); while (cursor.nextMatch()) |match| { for (match.captures) |cap| { const syn = synFor(sel.query.captureNameForId(cap.index) orelse ""); if (syn == .none) continue; const b = @min(base + cap.node.startByte(), styles.len); const end = @min(base + @as(usize, cap.node.endByte()), styles.len); if (end > b) @memset(styles[b..end], @intFromEnum(syn)); } } } // Queries compile on first use (`ensure`), never at startup. Pre-compiling the // compact tier in Pardes.init cost EVERY boot ~120ms of ts_query__perform_analysis // (55% of a Debug startup) to save ~40ms on the first .zig/.c/.cpp open — a pane // of prose or a shell paid for a language it never opened. Grammar availability // is unchanged; only the timing moved. fn synFor(name: []const u8) Syn { for ([_]struct { []const u8, Syn }{ .{ "comment", .comment }, .{ "string", .string }, .{ "character", .string }, .{ "number", .number }, .{ "numeric", .number }, .{ "float", .number }, .{ "boolean", .number }, .{ "keyword", .keyword }, .{ "include", .keyword }, .{ "conditional", .keyword }, .{ "repeat", .keyword }, .{ "title", .keyword }, .{ "uri", .string }, .{ "reference", .number }, // Markup grammars (markdown, typst) name prose constructs in their own // vocabulary rather than the code vocabulary above, so none of the // needles so far reach them. These are appended, and first-match-wins // makes that strictly additive; the needles below were audited across // all 27 shipped queries and occur only in the markdown and typst ones, // so no other language is recoloured. // // The slot assignment is forced by the palette being four wide and by // `synStyle` attaching the real BOLD attribute to exactly two of them, // `keyword` and `comment`: headings take `keyword`, so bold spans have // to land on `comment` to render actually bold. Nothing is left that // renders italic, so emphasis can only get a colour shift (`number`). .{ "heading", .keyword }, .{ "strong", .comment }, .{ "bold", .comment }, .{ "emphasis", .number }, .{ "italic", .number }, .{ "literal", .string }, .{ "raw", .string }, }) |m| { if (std.mem.indexOf(u8, name, m[0]) != null) return m[1]; } return .none; } /// One Syn byte per content byte in [start, end). Caller frees. pub fn highlightFileRange(gpa: std.mem.Allocator, path: []const u8, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { const tz = tracy.zone(@src(), "highlightFileRange"); defer tz.end(); if (!enabled) return &.{}; const ext = std.fs.path.extension(path); const selected = (forExt(ext) catch return &.{}) orelse return &.{}; const start_byte = @min(start_byte_raw, content.len); const end_byte = @max(start_byte, @min(end_byte_raw, content.len)); const source = content[start_byte..end_byte]; const styles = try gpa.alloc(u8, source.len); errdefer gpa.free(styles); @memset(styles, 0); const parser = ts.Parser.create(); defer parser.destroy(); parser.setLanguage(selected.lang) catch return styles; const cursor = ts.QueryCursor.create(); defer cursor.destroy(); paintWith(styles, source, selected, parser, cursor, true); return styles; } /// Parse `source` and write its style bytes into `styles`, with a parser and a /// query cursor the caller owns, so the per-row pass below can run a whole /// results buffer through one of each. /// /// `inject` is off for a single row. Both injection passes build a SECOND /// parser of their own — per fenced block, per `inline` node — which is /// amortised over a document and absurd over one truncated grep row that /// almost never contains a fenced block to begin with. fn paintWith( styles: []u8, source: []const u8, selected: Selected, parser: *ts.Parser, cursor: *ts.QueryCursor, inject: bool, ) void { const tree = parser.parseString(source, null) orelse return; defer tree.destroy(); runQuery(styles, selected, tree, 0, cursor); if (!inject) return; if (InjectSite.forGrammar(selected.name)) |site| injectCodeBlocks(styles, source, tree.rootNode(), site); // Disjoint from the fenced-block pass above: `code_fence_content` is never // an `inline` node, so the two never write the same byte. if (std.mem.eql(u8, selected.name, "markdown")) injectMarkdownInline(styles, source, tree.rootNode()); } /// A results buffer — every search, grep and language answer in this program — /// coloured as the CODE it is quoting. /// /// The rows look like `src/look.zig:718:12-16 fn grepText(path: []const u8...`: /// a location, a space, and a piece of some file. The location names the file, /// the file names the grammar, and the rest of the row is a fragment of that /// language — so a +Grep over Zig reads as Zig and one over Markdown does not /// pretend to. `look.parsePathLine` decides what counts as a location, which is /// the same primitive n/N walks these buffers with, so the two agree by /// construction about which rows are locations. /// /// ONE PARSER AND ONE CURSOR for the whole buffer, and the buffer is coloured /// once when it is filled rather than per visible window (file_pane /// `refreshHighlights`) — the rows are independent, so a window pass buys no /// fidelity and pays a burst of parses, cursors and first-time query compiles /// on every scroll that outran the covered range. pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { const tz = tracy.zone(@src(), "highlightLocations"); defer tz.end(); const start_byte = @min(start_byte_raw, content.len); const end_byte = @max(start_byte, @min(end_byte_raw, content.len)); const source = content[start_byte..end_byte]; if (!enabled) return &.{}; const styles = try gpa.alloc(u8, source.len); errdefer gpa.free(styles); @memset(styles, 0); // Both are created on the first row that needs them and kept for the rest; // `held` is the language the parser is currently set to. var parser: ?*ts.Parser = null; defer if (parser) |ptr| ptr.destroy(); var cursor: ?*ts.QueryCursor = null; defer if (cursor) |ptr| ptr.destroy(); var held: ?Selected = null; // Consecutive rows of a results buffer are overwhelmingly the same file, // and `forExt` is a linear walk of 29 specs and their extension lists. One // remembered answer collapses that to a string compare — including for the // rows that match NOTHING (a jumplist `@p3:10`, a `.lock`, a `.txt`), // which otherwise pay the whole failing scan every time. var memo_ext: []const u8 = "\x00"; var memo: ?Selected = null; var painted = false; var offset: usize = 0; var lines = std.mem.splitScalar(u8, source, '\n'); while (lines.next()) |line| { defer offset += line.len + 1; const code = codeAfterLocation(line) orelse continue; const ext = std.fs.path.extension(code.path); if (!std.mem.eql(u8, ext, memo_ext)) { memo_ext = ext; memo = forExt(ext) catch null; } const selected = memo orelse continue; if (parser == null) parser = ts.Parser.create(); if (cursor == null) cursor = ts.QueryCursor.create(); if (held == null or held.?.lang != selected.lang) { // `held` is cleared FIRST: a failed `setLanguage` has already set // the parser's language to null, so leaving `held` on the previous // grammar makes every later row of it skip the call and parse // against nothing — the rest of the buffer silently loses colour. held = null; parser.?.setLanguage(selected.lang) catch continue; held = selected; } paintWith(styles[offset + code.at ..][0..code.text.len], code.text, selected, parser.?, cursor.?, false); painted = true; } // NOTHING TO PAINT IS NOTHING TO KEEP. `recolorSyntax` skips a pane whose // highlights are empty, and every output buffer without locations in it — // +Help, +Config, +Messages, +Errors — would otherwise hand the renderer a // full-length run of zeroes and make it walk every visible grapheme, every // frame, to paint nothing. if (!painted) { gpa.free(styles); return &.{}; } return styles; } /// The ` ` split of one results row, or null when the row is not /// one. A row qualifies when its FIRST whitespace-delimited token is entirely a /// look target — the whole token, so `see:` in prose does not count — and /// something follows it. fn codeAfterLocation(line: []const u8) ?struct { path: []const u8, at: usize, text: []const u8 } { const token_end = std.mem.indexOfAny(u8, line, " \t") orelse return null; if (token_end == 0) return null; const token = line[0..token_end]; const target = look.parsePathLine(token); if (target.end != token.len) return null; // A bare word is not a location: `main.zig` alone is a filename, but a // results row is `main.zig:12:3`, and without that a prose line whose // first word happens to end in `.md` would colour the rest of a sentence. if (target.at.line == 0) return null; var at = token_end; while (at < line.len and (line[at] == ' ' or line[at] == '\t')) at += 1; if (at >= line.len) return null; return .{ .path = target.path, .at = at, .text = line[at..] }; } // Markdown fenced blocks and Typst raw blocks are the same construct — a // language tag plus a literal payload — under different node shapes, so one // walker drives both and only the (lang, content) extraction differs. const InjectSite = enum { markdown_fence, typst_raw, fn forGrammar(name: []const u8) ?InjectSite { if (std.mem.eql(u8, name, "markdown")) return .markdown_fence; if (std.mem.eql(u8, name, "typst")) return .typst_raw; return null; } fn blockKind(self: InjectSite) []const u8 { return switch (self) { .markdown_fence => "fenced_code_block", .typst_raw => "raw_blck", }; } /// null when the block carries no language tag (an untagged fence, or a /// Typst raw block written without one) — nothing to inject, leave it alone. fn parts(self: InjectSite, block: ts.Node) ?struct { lang: ts.Node, content: ts.Node } { switch (self) { .markdown_fence => { const info = childOfKind(block, "info_string") orelse return null; return .{ .lang = childOfKind(info, "language") orelse return null, .content = childOfKind(block, "code_fence_content") orelse return null, }; }, .typst_raw => return .{ .lang = block.childByFieldName("lang") orelse return null, .content = childOfKind(block, "blob") orelse return null, }, } } }; fn injectCodeBlocks(styles: []u8, source: []const u8, node: ts.Node, site: InjectSite) void { if (std.mem.eql(u8, node.kind(), site.blockKind())) { highlightCodeBlock(styles, source, node, site); return; } var i: u32 = 0; const count = node.childCount(); while (i < count) : (i += 1) { if (node.child(i)) |c| injectCodeBlocks(styles, source, c, site); } } fn childOfKind(node: ts.Node, kind: []const u8) ?ts.Node { var i: u32 = 0; const count = node.childCount(); while (i < count) : (i += 1) { if (node.child(i)) |c| { if (std.mem.eql(u8, c.kind(), kind)) return c; } } return null; } fn highlightCodeBlock(styles: []u8, source: []const u8, block: ts.Node, site: InjectSite) void { const p = site.parts(block) orelse return; const langtext = source[p.lang.startByte()..p.lang.endByte()]; const sub_sel = (forLang(langtext) catch return) orelse return; const cs: usize = p.content.startByte(); const ce: usize = p.content.endByte(); if (ce > source.len or cs > ce) return; const parser = ts.Parser.create(); defer parser.destroy(); parser.setLanguage(sub_sel.lang) catch return; const tree = parser.parseString(source[cs..ce], null) orelse return; defer tree.destroy(); // The outer grammar already painted these bytes (Typst blankets the whole // raw block `string`), and runQuery only writes bytes it captures, so the // outer colour would survive as a wash behind the injected code. The sub // grammar owns the payload outright: clear it first. @memset(styles[cs..ce], @intFromEnum(Syn.none)); const cursor = ts.QueryCursor.create(); defer cursor.destroy(); runQuery(styles, sub_sel, tree, cs, cursor); } // tree-sitter-markdown is a split grammar: the block parser bottoms out at named // `inline` nodes whose bytes it never looks inside, and emphasis / // strong_emphasis / code_span exist only in the companion inline parser. So // every `inline` node is re-parsed with `markdown_inline`, which is what // upstream's injections.scm, helix and nvim all do (per node, uncombined — the // stray block_continuation markers inside a multi-line paragraph's range are // just plain text to the inline parser). // // The grammar is resolved and the parser built once per file, not once per node: // only the parse is inherently per node. fn injectMarkdownInline(styles: []u8, source: []const u8, root: ts.Node) void { // Absent in builds below the `full` tier — nothing to inject, leave the // block grammar's colours alone. const sub_sel = (forLang("markdown_inline") catch return) orelse return; const parser = ts.Parser.create(); defer parser.destroy(); parser.setLanguage(sub_sel.lang) catch return; inlineNodes(styles, source, root, sub_sel, parser); } fn inlineNodes(styles: []u8, source: []const u8, node: ts.Node, sel: Selected, parser: *ts.Parser) void { if (std.mem.eql(u8, node.kind(), "inline")) { highlightInline(styles, source, node, sel, parser); return; // `inline` nodes never nest } var i: u32 = 0; const count = node.childCount(); while (i < count) : (i += 1) { if (node.child(i)) |c| inlineNodes(styles, source, c, sel, parser); } } fn highlightInline(styles: []u8, source: []const u8, node: ts.Node, sel: Selected, parser: *ts.Parser) void { const s: usize = node.startByte(); const e: usize = node.endByte(); if (s >= e or e > source.len) return; // an empty atx heading has a zero-length inline const tree = parser.parseString(source[s..e], null) orelse return; defer tree.destroy(); // Deliberately NOT clearing the range first, unlike highlightCodeBlock: the // block query already painted a heading's inline text `keyword`, and that is // the colour the heading must keep wherever the inline pass captures // nothing. Painting over instead of resetting is what makes *italic* inside // a heading recolour while the rest of the heading stays heading-coloured. const cursor = ts.QueryCursor.create(); defer cursor.destroy(); runQuery(styles, sel, tree, s, cursor); } /// One Syn byte per byte of content[start, end) for unified diffs/patches. /// Pure byte scan; independent of tree-sitter and the `enabled` flag. Caller frees. pub fn highlightDiff(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { const start_byte = @min(start_byte_raw, content.len); const end_byte = @max(start_byte, @min(end_byte_raw, content.len)); const source = content[start_byte..end_byte]; const styles = try gpa.alloc(u8, source.len); errdefer gpa.free(styles); @memset(styles, 0); var offset: usize = 0; var lines = std.mem.splitScalar(u8, source, '\n'); while (lines.next()) |line| { const syn = diffLineSyn(line); if (syn != .none) @memset(styles[offset .. offset + line.len], @intFromEnum(syn)); offset += line.len + 1; } return styles; } fn diffLineSyn(line: []const u8) Syn { if (std.mem.startsWith(u8, line, "@@")) return .keyword; if (std.mem.startsWith(u8, line, "+++") or std.mem.startsWith(u8, line, "---") or std.mem.startsWith(u8, line, "diff ") or std.mem.startsWith(u8, line, "index ") or std.mem.startsWith(u8, line, "\\ No newline")) return .comment; if (line.len == 0) return .none; if (line[0] == '+') return .string; if (line[0] == '-') return .number; return .none; } test "a results row is coloured by the file its location names" { if (!enabled) return; start(std.testing.allocator); defer stop(); const gpa = std.testing.allocator; // Two rows quoting two languages, plus a row that is not a location and a // location with nothing after it. const content = "src/a.zig:1:1 const S = struct {};\n" ++ "src/b.md:2:1 # heading\n" ++ "just some prose with a colon: here\n" ++ "src/c.zig:3:1\n"; const styles = try highlightLocations(gpa, content, 0, content.len); defer gpa.free(styles); try std.testing.expectEqual(content.len, styles.len); // The LOCATION itself is left alone — it is not code, and colouring it as // code is how a path starts looking like a keyword. for (styles[0.."src/a.zig:1:1".len]) |b| try std.testing.expectEqual(@as(u8, 0), b); // ...and `struct` in the Zig row is a keyword, which is only true if the // grammar was chosen from `a.zig` rather than from the buffer's own name. // // `struct` and not `const`: tree-sitter-zig captures `const` as // `@type.qualifier`, which `synFor` maps to nothing — a real property of // the shipped query rather than of this pass, and the reason the first // version of this test failed. const zig_kw = std.mem.indexOf(u8, content, "struct").?; try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw]); try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw + 5]); // A row with no location contributes nothing... const prose = std.mem.indexOf(u8, content, "just some prose").?; for (styles[prose .. prose + 14]) |b| try std.testing.expectEqual(@as(u8, 0), b); // ...and neither does a location with no code after it. const bare = std.mem.indexOf(u8, content, "src/c.zig").?; for (styles[bare..]) |b| try std.testing.expectEqual(@as(u8, 0), b); } test "a buffer with no locations in it keeps no highlights at all" { if (!enabled) return; start(std.testing.allocator); defer stop(); // +Help, +Config, +Messages: prose. An all-zero run of styles is not the // same as none — `recolorSyntax` skips a pane whose highlights are EMPTY, // and returning a full-length run of zeroes made it walk every visible // grapheme every frame to paint nothing. const content = "nothing has been said yet\n0: save: AccessDenied (x2)\n"; const styles = try highlightLocations(std.testing.allocator, content, 0, content.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(@as(usize, 0), styles.len); } test "codeAfterLocation takes whole-token locations and nothing else" { // A grep row: path, line, column range, then the quoted source. const got = codeAfterLocation("src/x.zig:7:2-9 fn main() void {") orelse return error.ShouldBeALocation; try std.testing.expectEqualStrings("src/x.zig", got.path); try std.testing.expectEqualStrings("fn main() void {", got.text); // Not locations: a bare filename (no line), prose with a colon, a token // that only PARTLY parses, and a location with nothing after it. try std.testing.expect(codeAfterLocation("main.zig some words") == null); try std.testing.expect(codeAfterLocation("note: this is prose") == null); try std.testing.expect(codeAfterLocation("src/x.zig:7:2x rest") == null); try std.testing.expect(codeAfterLocation("src/x.zig:7:2") == null); try std.testing.expect(codeAfterLocation("") == null); try std.testing.expect(codeAfterLocation(" leading space") == null); } test "tree-sitter allocator callbacks preserve and free exact allocations" { syntax_allocator = std.testing.allocator; defer syntax_allocator = undefined; var live: ?*anyopaque = syntaxCalloc(4, 1) orelse return error.OutOfMemory; defer if (live) |pointer| syntaxFree(pointer); const original: [*]u8 = @ptrCast(live.?); try std.testing.expectEqualSlices(u8, &.{ 0, 0, 0, 0 }, original[0..4]); @memcpy(original[0..4], "data"); live = syntaxRealloc(live, 32) orelse return error.OutOfMemory; const grown: [*]u8 = @ptrCast(live.?); try std.testing.expectEqualSlices(u8, "data", grown[0..4]); try std.testing.expect(syntaxCalloc(std.math.maxInt(usize), 2) == null); try std.testing.expect(syntaxRealloc(live, 0) == null); live = null; } test "default full grammar set highlights Typst source" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = "// note\n#let answer = 42\n#let text = \"hello\"\n"; const styles = try highlightFileRange(std.testing.allocator, "paper.typst", source, 0, source.len); defer std.testing.allocator.free(styles); const comment_at = std.mem.indexOf(u8, source, "// note").?; const keyword_at = std.mem.indexOf(u8, source, "let").?; const number_at = std.mem.indexOf(u8, source, "42").?; const string_at = std.mem.indexOf(u8, source, "\"hello\"").?; try std.testing.expectEqual(Syn.comment, @as(Syn, @enumFromInt(styles[comment_at]))); try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(styles[keyword_at]))); try std.testing.expectEqual(Syn.number, @as(Syn, @enumFromInt(styles[number_at]))); try std.testing.expectEqual(Syn.string, @as(Syn, @enumFromInt(styles[string_at]))); const short_ext = try highlightFileRange(std.testing.allocator, "paper.typ", source, 0, source.len); defer std.testing.allocator.free(short_ext); try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(short_ext[keyword_at]))); } test "Typst markup constructs paint and raw blocks inject their language" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = \\// note \\= Heading Title \\Some *bold* text with `raw` inline. \\#import "mod.typ": helper \\#let answer = 42 \\#emit(7, "arg") \\```zig \\fn widget() u8 { return 99; } \\``` \\ ; const styles = try highlightFileRange(std.testing.allocator, "paper.typ", source, 0, source.len); defer std.testing.allocator.free(styles); const synAt = struct { fn f(s: []const u8, src: []const u8, needle: []const u8, offset: usize) Syn { return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]); } }.f; // marker and text of the heading both take the bold accent try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "= Heading", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Heading", 0)); // *bold* is @markup.bold, which lands on the one slot that still renders // with the real bold attribute now that headings own `keyword`. try std.testing.expectEqual(Syn.comment, synAt(styles, source, "bold", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "raw` inline", 0)); // the callee and the code sigil in front of it try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "emit", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "#emit", 0)); // fence and lang tag keep the literal colour of the raw block... try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 3)); // ...while the blob is reset and re-painted by the injected zig grammar. Its // `fn` and `99` prove the injection ran; `widget` proves the reset, since the // zig query names it @function (Syn.none here) and without clearing the blob // first the raw block's `string` would still be washing over it. try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn widget", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "99", 0)); try std.testing.expectEqual(Syn.none, synAt(styles, source, "widget", 0)); try std.testing.expectEqual(Syn.comment, synAt(styles, source, "// note", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "import", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "let", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "42", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "\"arg\"", 0)); } test "markdown highlights markup, injects inline spans and fenced code blocks" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = "# Heading *slant* Title\n" ++ "\n" ++ "Prose with **stout** and *lean* plus `snippet` inline.\n" ++ "\n" ++ "```zig\n" ++ "fn gadget() u8 { return 77; }\n" ++ "```\n"; const styles = try highlightFileRange(std.testing.allocator, "doc.md", source, 0, source.len); defer std.testing.allocator.free(styles); const synAt = struct { fn f(s: []const u8, src: []const u8, needle: []const u8, offset: usize) Syn { return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]); } }.f; // The block grammar washes the whole heading `keyword`; the inline pass then // paints the emphasis over it without resetting, so the heading keeps its // colour everywhere the emphasis is not. try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Title", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "slant", 0)); // bold/italic/code-span live only in the inline grammar, so all three prove // the inline injection ran. try std.testing.expectEqual(Syn.comment, synAt(styles, source, "stout", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "lean", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "snippet", 0)); // the fence is inside the block query's @text.literal wash... try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0)); // ...while the payload is cleared and re-painted by the injected zig // grammar: `fn` and `77` prove the injection ran, and `gadget` proves the // clear, since the zig query names it @function (Syn.none here) and without // clearing first the fence's `string` would still be washing over it. try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn gadget", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "77", 0)); try std.testing.expectEqual(Syn.none, synAt(styles, source, "gadget", 0)); } test "highlightDiff colors unified diff lines by prefix" { const diff = "diff --git a/x b/x\n" ++ "--- a/x\n" ++ "+++ b/x\n" ++ "@@ -1,3 +1,3 @@\n" ++ " context\n" ++ "-old line\n" ++ "+new line\n"; const styles = try highlightDiff(std.testing.allocator, diff, 0, diff.len); defer std.testing.allocator.free(styles); const byteSyn = struct { fn at(s: []const u8, src: []const u8, needle: []const u8) Syn { const i = std.mem.indexOf(u8, src, needle).?; return @enumFromInt(s[i]); } }.at; try std.testing.expectEqual(Syn.comment, byteSyn(styles, diff, "diff --git")); try std.testing.expectEqual(Syn.comment, byteSyn(styles, diff, "--- a/x")); try std.testing.expectEqual(Syn.comment, byteSyn(styles, diff, "+++ b/x")); try std.testing.expectEqual(Syn.keyword, byteSyn(styles, diff, "@@ -1,3")); try std.testing.expectEqual(Syn.none, byteSyn(styles, diff, " context")); try std.testing.expectEqual(Syn.number, byteSyn(styles, diff, "-old line")); try std.testing.expectEqual(Syn.string, byteSyn(styles, diff, "+new line")); }