diff options
Diffstat (limited to 'src/syntax.zig')
| -rw-r--r-- | src/syntax.zig | 651 |
1 files changed, 320 insertions, 331 deletions
diff --git a/src/syntax.zig b/src/syntax.zig index f6e7cae2..77ef80d7 100644 --- a/src/syntax.zig +++ b/src/syntax.zig @@ -1,7 +1,3 @@ -//! Tree-sitter syntax highlighting: one style byte per content byte, filled by -//! running each grammar's highlights.scm query (slurped at build time into the -//! ts_queries options module). Grammar set is tiered: `zig`; `minimal` (c, cpp, -//! zig); and `full`, which adds ~24 languages lazily on first use. const std = @import("std"); const config = @import("pardes_config"); const tracy = @import("tracy.zig"); @@ -16,6 +12,8 @@ const full_grammars = config.syntax_full_grammars; const ts = if (enabled) @import("tree-sitter") else struct { pub const Language = opaque {}; pub const Query = opaque {}; + pub const Parser = opaque {}; + pub const QueryCursor = opaque {}; }; const ts_queries = if (enabled) @import("ts_queries") else struct {}; @@ -28,7 +26,8 @@ const Spec = struct { exts: []const []const u8, language: *const fn () callconv(.c) *const ts.Language, query_src: []const u8, - compiled_query: ?*ts.Query = null, + selected: ?Selected = null, + capture_styles: [256]u8 = undefined, }; fn grammarSelected(comptime g: grammar_manifest.Grammar) bool { @@ -46,21 +45,6 @@ fn specCount() comptime_int { } return count; } - -// Upstream's typst highlights.scm names its markup honestly — -// @markup.heading.*, @markup.bold, @markup.italic, @markup.raw.block — and -// `synFor` now understands that vocabulary, so headings, bold, italics and raw -// blocks paint on their own. What is left here is exactly the two captures we -// refuse to map globally: -// -// - upstream tags call callees @function/@function.method; mapping "function" -// in `synFor` would recolour every function call in every language. -// - upstream tags the code sigil "#" @operator, and mapping "operator" would -// likewise light up every +, -, == in the codebase. -// -// Both are worth colouring *in typst specifically*: the sigil in front of every -// #let/#if/#import/#call is the visual anchor of the code/markup split, and -// upstream leaves it uncoloured even though the keyword behind it is not. const typst_supplement = \\ \\(call item: (ident) @keyword) @@ -71,8 +55,69 @@ const typst_supplement = fn querySrc(comptime g: grammar_manifest.Grammar) []const u8 { const base = @field(ts_queries, g.name ++ "_highlights"); - if (comptime std.mem.eql(u8, g.name, "typst")) return base ++ typst_supplement; - return base; + const source = if (comptime std.mem.eql(u8, g.name, "typst")) base ++ typst_supplement else base; + return comptime colorQuery(source); +} + +fn colorQuery(comptime source: []const u8) []const u8 { + @setEvalBranchQuota(2_000_000); + var result: [source.len]u8 = undefined; + var written: usize = 0; + var at: usize = 0; + while (at < source.len) { + while (at < source.len) { + if (std.ascii.isWhitespace(source[at])) { + at += 1; + } else if (source[at] == ';') { + while (at < source.len and source[at] != '\n') at += 1; + } else break; + } + const begin = at; + var depth: usize = 0; + var colored = false; + var complete = false; + while (at < source.len) { + const byte = source[at]; + if (complete and (byte == '(' or byte == '[' or byte == '"')) break; + if (byte == ';') { + while (at < source.len and source[at] != '\n') at += 1; + continue; + } + if (byte == '"') { + at += 1; + while (at < source.len) : (at += 1) { + if (source[at] == '\\') { + at += 1; + } else if (source[at] == '"') { + at += 1; + break; + } + } + if (depth == 0) complete = true; + continue; + } + if (byte == '(' or byte == '[') depth += 1; + if (byte == ')' or byte == ']') { + depth -= 1; + if (depth == 0) complete = true; + } + if (byte == '@') { + const name = at + 1; + at = name; + while (at < source.len and (std.ascii.isAlphanumeric(source[at]) or + source[at] == '_' or source[at] == '.' or source[at] == '-')) at += 1; + colored = colored or synFor(source[name..at]) != .none; + continue; + } + at += 1; + } + if (colored) { + @memcpy(result[written..][0 .. at - begin], source[begin..at]); + written += at - begin; + } + } + const filtered = result[0..written].*; + return &filtered; } fn initSpecs() [specCount()]Spec { @@ -151,6 +196,7 @@ fn syntaxFree(ptr: ?*anyopaque) callconv(.c) void { pub fn start(gpa: std.mem.Allocator) void { if (comptime enabled) { std.debug.assert(!syntax_started); + stop(); syntax_allocator = gpa; syntax_started = true; ts_set_allocator(syntaxAlloc, syntaxCalloc, syntaxRealloc, syntaxFree); @@ -159,10 +205,13 @@ pub fn start(gpa: std.mem.Allocator) void { pub fn stop() void { if (comptime enabled) { - std.debug.assert(syntax_started); for (&specs) |*spec| { - if (spec.compiled_query) |query| query.destroy(); - spec.compiled_query = null; + if (spec.selected) |selected| { + selected.cursor.destroy(); + selected.parser.destroy(); + selected.query.destroy(); + } + spec.selected = null; } ts_set_allocator(null, null, null, null); syntax_started = false; @@ -170,17 +219,41 @@ pub fn stop() void { } } -const Selected = struct { name: []const u8, lang: *const ts.Language, query: *ts.Query }; +const Selected = struct { + name: []const u8, + lang: *const ts.Language, + query: *ts.Query, + parser: *ts.Parser, + cursor: *ts.QueryCursor, + capture_styles: []u8, +}; -// NOTE: don't lang.destroy() — the tree_sitter_*() languages are static -// singletons reused on every open; destroying one use-after-frees the next. fn ensure(spec: *Spec) !Selected { + if (spec.selected) |selected| return selected; const lang = spec.language(); - if (spec.compiled_query) |query| return .{ .name = spec.name, .lang = lang, .query = query }; var error_offset: u32 = 0; const query = try ts.Query.create(lang, spec.query_src, &error_offset); - spec.compiled_query = query; - return .{ .name = spec.name, .lang = lang, .query = query }; + errdefer query.destroy(); + if (query.captureCount() > spec.capture_styles.len) return error.TooManyCaptures; + const capture_styles = spec.capture_styles[0..query.captureCount()]; + for (capture_styles, 0..) |*style, id| { + const name = query.captureNameForId(@intCast(id)) orelse ""; + style.* = @intFromEnum(synFor(name)); + if (style.* == 0) query.disableCapture(name); + } + const parser = ts.Parser.create(); + errdefer parser.destroy(); + try parser.setLanguage(lang); + const selected: Selected = .{ + .name = spec.name, + .lang = lang, + .query = query, + .parser = parser, + .cursor = ts.QueryCursor.create(), + .capture_styles = capture_styles, + }; + spec.selected = selected; + return selected; } fn forExt(ext: []const u8) !?Selected { @@ -226,30 +299,20 @@ fn forLang(name: []const u8) !?Selected { } return null; } - -/// The caller owns the cursor: `ts_query_cursor_exec` fully resets its state, -/// so one cursor serves any number of trees, and the per-row pass would -/// otherwise create and destroy one — three allocations against the shared -/// tree-sitter arena — for every row of a results buffer. -fn runQuery(styles: []u8, sel: Selected, tree: *ts.Tree, base: usize, cursor: *ts.QueryCursor) void { +fn runQuery(styles: []u8, sel: Selected, tree: *ts.Tree, base: usize) void { + const cursor = sel.cursor; cursor.exec(sel.query, tree.rootNode()); while (cursor.nextMatch()) |match| { for (match.captures) |cap| { - const syn = synFor(sel.query.captureNameForId(cap.index) orelse ""); - if (syn == .none) continue; + const style = sel.capture_styles[cap.index]; + if (style == 0) continue; const b = @min(base + cap.node.startByte(), styles.len); const end = @min(base + @as(usize, cap.node.endByte()), styles.len); - if (end > b) @memset(styles[b..end], @intFromEnum(syn)); + if (end > b) @memset(styles[b..end], style); } } } -// Queries compile on first use (`ensure`), never at startup. Pre-compiling the -// compact tier in Pardes.init cost EVERY boot ~120ms of ts_query__perform_analysis -// (55% of a Debug startup) to save ~40ms on the first .zig/.c/.cpp open — a pane -// of prose or a shell paid for a language it never opened. Grammar availability -// is unchanged; only the timing moved. - fn synFor(name: []const u8) Syn { for ([_]struct { []const u8, Syn }{ .{ "comment", .comment }, @@ -266,19 +329,6 @@ fn synFor(name: []const u8) Syn { .{ "title", .keyword }, .{ "uri", .string }, .{ "reference", .number }, - - // Markup grammars (markdown, typst) name prose constructs in their own - // vocabulary rather than the code vocabulary above, so none of the - // needles so far reach them. These are appended, and first-match-wins - // makes that strictly additive; the needles below were audited across - // all 27 shipped queries and occur only in the markdown and typst ones, - // so no other language is recoloured. - // - // The slot assignment is forced by the palette being four wide and by - // `synStyle` attaching the real BOLD attribute to exactly two of them, - // `keyword` and `comment`: headings take `keyword`, so bold spans have - // to land on `comment` to render actually bold. Nothing is left that - // renders italic, so emphasis can only get a colour shift (`number`). .{ "heading", .keyword }, .{ "strong", .comment }, .{ "bold", .comment }, @@ -291,75 +341,31 @@ fn synFor(name: []const u8) Syn { } return .none; } - -/// One Syn byte per content byte in [start, end). Caller frees. -pub fn highlightFileRange(gpa: std.mem.Allocator, path: []const u8, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { - const tz = tracy.zone(@src(), "highlightFileRange"); - defer tz.end(); +pub fn highlightFileRange(gpa: std.mem.Allocator, path: []const u8, content: []const u8, start_raw: usize, end_raw: usize) ![]u8 { + const zone = tracy.zone(@src(), "highlightFileRange"); + defer zone.end(); if (!enabled) return &.{}; - const ext = std.fs.path.extension(path); - const selected = (forExt(ext) catch return &.{}) orelse return &.{}; - - const start_byte = @min(start_byte_raw, content.len); - const end_byte = @max(start_byte, @min(end_byte_raw, content.len)); + const selected = (try forExt(std.fs.path.extension(path))) orelse return &.{}; + const start_byte = @min(start_raw, content.len); + const end_byte = @max(start_byte, @min(end_raw, content.len)); const source = content[start_byte..end_byte]; const styles = try gpa.alloc(u8, source.len); - errdefer gpa.free(styles); @memset(styles, 0); - - const parser = ts.Parser.create(); - defer parser.destroy(); - parser.setLanguage(selected.lang) catch return styles; - const cursor = ts.QueryCursor.create(); - defer cursor.destroy(); - paintWith(styles, source, selected, parser, cursor, true); + paint(styles, source, selected); return styles; } -/// Parse `source` and write its style bytes into `styles`, with a parser and a -/// query cursor the caller owns, so the per-row pass below can run a whole -/// results buffer through one of each. -/// -/// `inject` is off for a single row. Both injection passes build a SECOND -/// parser of their own — per fenced block, per `inline` node — which is -/// amortised over a document and absurd over one truncated grep row that -/// almost never contains a fenced block to begin with. -fn paintWith( - styles: []u8, - source: []const u8, - selected: Selected, - parser: *ts.Parser, - cursor: *ts.QueryCursor, - inject: bool, -) void { - const tree = parser.parseString(source, null) orelse return; +fn paint(styles: []u8, source: []const u8, selected: Selected) void { + const tree = selected.parser.parseString(source, null) orelse return; defer tree.destroy(); - - runQuery(styles, selected, tree, 0, cursor); - if (!inject) return; - - if (InjectSite.forGrammar(selected.name)) |site| injectCodeBlocks(styles, source, tree.rootNode(), site); - // Disjoint from the fenced-block pass above: `code_fence_content` is never - // an `inline` node, so the two never write the same byte. - if (std.mem.eql(u8, selected.name, "markdown")) injectMarkdownInline(styles, source, tree.rootNode()); + runQuery(styles, selected, tree, 0); + if (std.mem.eql(u8, selected.name, "markdown")) { + inject(styles, source, tree.rootNode(), true); + } else if (std.mem.eql(u8, selected.name, "typst")) { + inject(styles, source, tree.rootNode(), false); + } } -/// A results buffer — every search, grep and language answer in this program — -/// coloured as the CODE it is quoting. -/// -/// The rows look like `src/look.zig:718:12-16 fn grepText(path: []const u8...`: -/// a location, a space, and a piece of some file. The location names the file, -/// the file names the grammar, and the rest of the row is a fragment of that -/// language — so a +Grep over Zig reads as Zig and one over Markdown does not -/// pretend to. `look.parsePathLine` decides what counts as a location, which is -/// the same primitive n/N walks these buffers with, so the two agree by -/// construction about which rows are locations. -/// -/// ONE PARSER AND ONE CURSOR for the whole buffer, and the buffer is coloured -/// once when it is filled rather than per visible window (file_pane -/// `refreshHighlights`) — the rows are independent, so a window pass buys no -/// fidelity and pays a burst of parses, cursors and first-time query compiles -/// on every scroll that outran the covered range. pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { const tz = tracy.zone(@src(), "highlightLocations"); defer tz.end(); @@ -370,19 +376,6 @@ pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byt const styles = try gpa.alloc(u8, source.len); errdefer gpa.free(styles); @memset(styles, 0); - - // Both are created on the first row that needs them and kept for the rest; - // `held` is the language the parser is currently set to. - var parser: ?*ts.Parser = null; - defer if (parser) |ptr| ptr.destroy(); - var cursor: ?*ts.QueryCursor = null; - defer if (cursor) |ptr| ptr.destroy(); - var held: ?Selected = null; - // Consecutive rows of a results buffer are overwhelmingly the same file, - // and `forExt` is a linear walk of 29 specs and their extension lists. One - // remembered answer collapses that to a string compare — including for the - // rows that match NOTHING (a jumplist `@p3:10`, a `.lock`, a `.txt`), - // which otherwise pay the whole failing scan every time. var memo_ext: []const u8 = "\x00"; var memo: ?Selected = null; var painted = false; @@ -398,100 +391,61 @@ pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byt memo = forExt(ext) catch null; } const selected = memo orelse continue; - if (parser == null) parser = ts.Parser.create(); - if (cursor == null) cursor = ts.QueryCursor.create(); - if (held == null or held.?.lang != selected.lang) { - // `held` is cleared FIRST: a failed `setLanguage` has already set - // the parser's language to null, so leaving `held` on the previous - // grammar makes every later row of it skip the call and parse - // against nothing — the rest of the buffer silently loses colour. - held = null; - parser.?.setLanguage(selected.lang) catch continue; - held = selected; - } - paintWith(styles[offset + code.at ..][0..code.text.len], code.text, selected, parser.?, cursor.?, false); + paint(styles[offset + code.at ..][0..code.text.len], code.text, selected); painted = true; } - // NOTHING TO PAINT IS NOTHING TO KEEP. `recolorSyntax` skips a pane whose - // highlights are empty, and every output buffer without locations in it — - // +Help, +Config, +Messages, +Errors — would otherwise hand the renderer a - // full-length run of zeroes and make it walk every visible grapheme, every - // frame, to paint nothing. if (!painted) { gpa.free(styles); return &.{}; } return styles; } - -/// The `<path> <code>` split of one results row, or null when the row is not -/// one. A row qualifies when its FIRST whitespace-delimited token is entirely a -/// look target — the whole token, so `see:` in prose does not count — and -/// something follows it. fn codeAfterLocation(line: []const u8) ?struct { path: []const u8, at: usize, text: []const u8 } { const token_end = std.mem.indexOfAny(u8, line, " \t") orelse return null; if (token_end == 0) return null; const token = line[0..token_end]; const target = look.parsePathLine(token); if (target.end != token.len) return null; - // A bare word is not a location: `main.zig` alone is a filename, but a - // results row is `main.zig:12:3`, and without that a prose line whose - // first word happens to end in `.md` would colour the rest of a sentence. if (target.at.line == 0) return null; - var at = token_end; - while (at < line.len and (line[at] == ' ' or line[at] == '\t')) at += 1; + const at = token_end + 1; if (at >= line.len) return null; return .{ .path = target.path, .at = at, .text = line[at..] }; } - -// Markdown fenced blocks and Typst raw blocks are the same construct — a -// language tag plus a literal payload — under different node shapes, so one -// walker drives both and only the (lang, content) extraction differs. -const InjectSite = enum { - markdown_fence, - typst_raw, - - fn forGrammar(name: []const u8) ?InjectSite { - if (std.mem.eql(u8, name, "markdown")) return .markdown_fence; - if (std.mem.eql(u8, name, "typst")) return .typst_raw; - return null; - } - - fn blockKind(self: InjectSite) []const u8 { - return switch (self) { - .markdown_fence => "fenced_code_block", - .typst_raw => "raw_blck", - }; - } - - /// null when the block carries no language tag (an untagged fence, or a - /// Typst raw block written without one) — nothing to inject, leave it alone. - fn parts(self: InjectSite, block: ts.Node) ?struct { lang: ts.Node, content: ts.Node } { - switch (self) { - .markdown_fence => { - const info = childOfKind(block, "info_string") orelse return null; - return .{ - .lang = childOfKind(info, "language") orelse return null, - .content = childOfKind(block, "code_fence_content") orelse return null, - }; - }, - .typst_raw => return .{ - .lang = block.childByFieldName("lang") orelse return null, - .content = childOfKind(block, "blob") orelse return null, - }, - } +fn inject(styles: []u8, source: []const u8, node: ts.Node, markdown: bool) void { + const kind = node.kind(); + if (markdown and std.mem.eql(u8, kind, "inline")) { + const begin: usize = node.startByte(); + const end: usize = node.endByte(); + if (begin >= end or end > source.len) return; + const text = source[begin..end]; + // Every colored inline capture requires one of these delimiters. + if (std.mem.indexOfAny(u8, text, "*_`[<\\\r\n") == null) return; + const selected = (forLang("markdown_inline") catch return) orelse return; + const tree = selected.parser.parseString(text, null) orelse return; + defer tree.destroy(); + runQuery(styles, selected, tree, begin); + return; } -}; - -fn injectCodeBlocks(styles: []u8, source: []const u8, node: ts.Node, site: InjectSite) void { - if (std.mem.eql(u8, node.kind(), site.blockKind())) { - highlightCodeBlock(styles, source, node, site); + if (std.mem.eql(u8, kind, if (markdown) "fenced_code_block" else "raw_blck")) { + const lang = if (markdown) blk: { + const info = childOfKind(node, "info_string") orelse return; + break :blk childOfKind(info, "language") orelse return; + } else node.childByFieldName("lang") orelse return; + const content = childOfKind(node, if (markdown) "code_fence_content" else "blob") orelse return; + const selected = (forLang(source[lang.startByte()..lang.endByte()]) catch return) orelse return; + const begin: usize = content.startByte(); + const end: usize = content.endByte(); + if (begin > end or end > source.len) return; + const tree = selected.parser.parseString(source[begin..end], null) orelse return; + defer tree.destroy(); + @memset(styles[begin..end], 0); + runQuery(styles, selected, tree, begin); return; } var i: u32 = 0; const count = node.childCount(); while (i < count) : (i += 1) { - if (node.child(i)) |c| injectCodeBlocks(styles, source, c, site); + if (node.child(i)) |child| inject(styles, source, child, markdown); } } @@ -506,79 +460,6 @@ fn childOfKind(node: ts.Node, kind: []const u8) ?ts.Node { return null; } -fn highlightCodeBlock(styles: []u8, source: []const u8, block: ts.Node, site: InjectSite) void { - const p = site.parts(block) orelse return; - const langtext = source[p.lang.startByte()..p.lang.endByte()]; - const sub_sel = (forLang(langtext) catch return) orelse return; - const cs: usize = p.content.startByte(); - const ce: usize = p.content.endByte(); - if (ce > source.len or cs > ce) return; - - const parser = ts.Parser.create(); - defer parser.destroy(); - parser.setLanguage(sub_sel.lang) catch return; - const tree = parser.parseString(source[cs..ce], null) orelse return; - defer tree.destroy(); - // The outer grammar already painted these bytes (Typst blankets the whole - // raw block `string`), and runQuery only writes bytes it captures, so the - // outer colour would survive as a wash behind the injected code. The sub - // grammar owns the payload outright: clear it first. - @memset(styles[cs..ce], @intFromEnum(Syn.none)); - const cursor = ts.QueryCursor.create(); - defer cursor.destroy(); - runQuery(styles, sub_sel, tree, cs, cursor); -} - -// tree-sitter-markdown is a split grammar: the block parser bottoms out at named -// `inline` nodes whose bytes it never looks inside, and emphasis / -// strong_emphasis / code_span exist only in the companion inline parser. So -// every `inline` node is re-parsed with `markdown_inline`, which is what -// upstream's injections.scm, helix and nvim all do (per node, uncombined — the -// stray block_continuation markers inside a multi-line paragraph's range are -// just plain text to the inline parser). -// -// The grammar is resolved and the parser built once per file, not once per node: -// only the parse is inherently per node. -fn injectMarkdownInline(styles: []u8, source: []const u8, root: ts.Node) void { - // Absent in builds below the `full` tier — nothing to inject, leave the - // block grammar's colours alone. - const sub_sel = (forLang("markdown_inline") catch return) orelse return; - const parser = ts.Parser.create(); - defer parser.destroy(); - parser.setLanguage(sub_sel.lang) catch return; - inlineNodes(styles, source, root, sub_sel, parser); -} - -fn inlineNodes(styles: []u8, source: []const u8, node: ts.Node, sel: Selected, parser: *ts.Parser) void { - if (std.mem.eql(u8, node.kind(), "inline")) { - highlightInline(styles, source, node, sel, parser); - return; // `inline` nodes never nest - } - var i: u32 = 0; - const count = node.childCount(); - while (i < count) : (i += 1) { - if (node.child(i)) |c| inlineNodes(styles, source, c, sel, parser); - } -} - -fn highlightInline(styles: []u8, source: []const u8, node: ts.Node, sel: Selected, parser: *ts.Parser) void { - const s: usize = node.startByte(); - const e: usize = node.endByte(); - if (s >= e or e > source.len) return; // an empty atx heading has a zero-length inline - const tree = parser.parseString(source[s..e], null) orelse return; - defer tree.destroy(); - // Deliberately NOT clearing the range first, unlike highlightCodeBlock: the - // block query already painted a heading's inline text `keyword`, and that is - // the colour the heading must keep wherever the inline pass captures - // nothing. Painting over instead of resetting is what makes *italic* inside - // a heading recolour while the rest of the heading stays heading-coloured. - const cursor = ts.QueryCursor.create(); - defer cursor.destroy(); - runQuery(styles, sel, tree, s, cursor); -} - -/// One Syn byte per byte of content[start, end) for unified diffs/patches. -/// Pure byte scan; independent of tree-sitter and the `enabled` flag. Caller frees. pub fn highlightDiff(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { const start_byte = @min(start_byte_raw, content.len); const end_byte = @max(start_byte, @min(end_byte_raw, content.len)); @@ -610,14 +491,11 @@ fn diffLineSyn(line: []const u8) Syn { return .none; } -test "a results row is coloured by the file its location names" { +test "syntax a results row is coloured by the file its location names" { if (!enabled) return; start(std.testing.allocator); defer stop(); const gpa = std.testing.allocator; - - // Two rows quoting two languages, plus a row that is not a location and a - // location with nothing after it. const content = "src/a.zig:1:1 const S = struct {};\n" ++ "src/b.md:2:1 # heading\n" ++ @@ -626,53 +504,31 @@ test "a results row is coloured by the file its location names" { const styles = try highlightLocations(gpa, content, 0, content.len); defer gpa.free(styles); try std.testing.expectEqual(content.len, styles.len); - - // The LOCATION itself is left alone — it is not code, and colouring it as - // code is how a path starts looking like a keyword. for (styles[0.."src/a.zig:1:1".len]) |b| try std.testing.expectEqual(@as(u8, 0), b); - - // ...and `struct` in the Zig row is a keyword, which is only true if the - // grammar was chosen from `a.zig` rather than from the buffer's own name. - // - // `struct` and not `const`: tree-sitter-zig captures `const` as - // `@type.qualifier`, which `synFor` maps to nothing — a real property of - // the shipped query rather than of this pass, and the reason the first - // version of this test failed. const zig_kw = std.mem.indexOf(u8, content, "struct").?; try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw]); try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw + 5]); - - // A row with no location contributes nothing... const prose = std.mem.indexOf(u8, content, "just some prose").?; for (styles[prose .. prose + 14]) |b| try std.testing.expectEqual(@as(u8, 0), b); - // ...and neither does a location with no code after it. const bare = std.mem.indexOf(u8, content, "src/c.zig").?; for (styles[bare..]) |b| try std.testing.expectEqual(@as(u8, 0), b); } -test "a buffer with no locations in it keeps no highlights at all" { +test "syntax a buffer with no locations in it keeps no highlights at all" { if (!enabled) return; start(std.testing.allocator); defer stop(); - // +Help, +Config, +Messages: prose. An all-zero run of styles is not the - // same as none — `recolorSyntax` skips a pane whose highlights are EMPTY, - // and returning a full-length run of zeroes made it walk every visible - // grapheme every frame to paint nothing. const content = "nothing has been said yet\n0: save: AccessDenied (x2)\n"; const styles = try highlightLocations(std.testing.allocator, content, 0, content.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(@as(usize, 0), styles.len); } -test "codeAfterLocation takes whole-token locations and nothing else" { - // A grep row: path, line, column range, then the quoted source. +test "syntax codeAfterLocation takes whole-token locations and nothing else" { const got = codeAfterLocation("src/x.zig:7:2-9 fn main() void {") orelse return error.ShouldBeALocation; try std.testing.expectEqualStrings("src/x.zig", got.path); try std.testing.expectEqualStrings("fn main() void {", got.text); - - // Not locations: a bare filename (no line), prose with a colon, a token - // that only PARTLY parses, and a location with nothing after it. try std.testing.expect(codeAfterLocation("main.zig some words") == null); try std.testing.expect(codeAfterLocation("note: this is prose") == null); try std.testing.expect(codeAfterLocation("src/x.zig:7:2x rest") == null); @@ -681,7 +537,7 @@ test "codeAfterLocation takes whole-token locations and nothing else" { try std.testing.expect(codeAfterLocation(" leading space") == null); } -test "tree-sitter allocator callbacks preserve and free exact allocations" { +test "syntax tree-sitter allocator callbacks preserve and free exact allocations" { syntax_allocator = std.testing.allocator; defer syntax_allocator = undefined; @@ -700,7 +556,7 @@ test "tree-sitter allocator callbacks preserve and free exact allocations" { live = null; } -test "default full grammar set highlights Typst source" { +test "syntax default full grammar set highlights Typst source" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); @@ -723,7 +579,7 @@ test "default full grammar set highlights Typst source" { try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(short_ext[keyword_at]))); } -test "Typst markup constructs paint and raw blocks inject their language" { +test "syntax Typst markup constructs paint and raw blocks inject their language" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); @@ -748,25 +604,14 @@ test "Typst markup constructs paint and raw blocks inject their language" { return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]); } }.f; - - // marker and text of the heading both take the bold accent try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "= Heading", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Heading", 0)); - // *bold* is @markup.bold, which lands on the one slot that still renders - // with the real bold attribute now that headings own `keyword`. try std.testing.expectEqual(Syn.comment, synAt(styles, source, "bold", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "raw` inline", 0)); - // the callee and the code sigil in front of it try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "emit", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "#emit", 0)); - - // fence and lang tag keep the literal colour of the raw block... try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 3)); - // ...while the blob is reset and re-painted by the injected zig grammar. Its - // `fn` and `99` prove the injection ran; `widget` proves the reset, since the - // zig query names it @function (Syn.none here) and without clearing the blob - // first the raw block's `string` would still be washing over it. try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn widget", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "99", 0)); try std.testing.expectEqual(Syn.none, synAt(styles, source, "widget", 0)); @@ -778,7 +623,7 @@ test "Typst markup constructs paint and raw blocks inject their language" { try std.testing.expectEqual(Syn.string, synAt(styles, source, "\"arg\"", 0)); } -test "markdown highlights markup, injects inline spans and fenced code blocks" { +test "syntax markdown highlights markup, injects inline spans and fenced code blocks" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); @@ -799,31 +644,80 @@ test "markdown highlights markup, injects inline spans and fenced code blocks" { return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]); } }.f; - - // The block grammar washes the whole heading `keyword`; the inline pass then - // paints the emphasis over it without resetting, so the heading keeps its - // colour everywhere the emphasis is not. try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Title", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "slant", 0)); - - // bold/italic/code-span live only in the inline grammar, so all three prove - // the inline injection ran. try std.testing.expectEqual(Syn.comment, synAt(styles, source, "stout", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "lean", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "snippet", 0)); - - // the fence is inside the block query's @text.literal wash... try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0)); - // ...while the payload is cleared and re-painted by the injected zig - // grammar: `fn` and `77` prove the injection ran, and `gadget` proves the - // clear, since the zig query names it @function (Syn.none here) and without - // clearing first the fence's `string` would still be washing over it. try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn gadget", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "77", 0)); try std.testing.expectEqual(Syn.none, synAt(styles, source, "gadget", 0)); } -test "highlightDiff colors unified diff lines by prefix" { +test "syntax inline fast path agrees with full Markdown query" { + if (!enabled or !full_grammars) return; + start(std.testing.allocator); + defer stop(); + const selected = (try forLang("markdown_inline")).?; + const check = struct { + fn compare(sel: Selected, source: []const u8) !void { + const tree = sel.parser.parseString(source, null) orelse return error.ParseFailed; + defer tree.destroy(); + const expected = try std.testing.allocator.alloc(u8, source.len); + defer std.testing.allocator.free(expected); + const actual = try std.testing.allocator.alloc(u8, source.len); + defer std.testing.allocator.free(actual); + for ([_]Syn{ .none, .keyword }) |background| { + @memset(expected, @intFromEnum(background)); + @memset(actual, @intFromEnum(background)); + runQuery(expected, sel, tree, 0); + inject(actual, source, tree.rootNode(), true); + if (!std.mem.eql(u8, expected, actual)) std.debug.print("inline mismatch: {s}\n", .{source}); + try std.testing.expectEqualSlices(u8, expected, actual); + } + } + }.compare; + for ([_][]const u8{ + "", "plain prose", + "ação Ελληνικά 日本語 🙂", + "123 456", "tabs\tand spaces", + "'quoted' (parentheses) \"double quotes\"", "https://example.org [email protected]", + "&  ", "~~struck~~ $formula$", + "*emphasis* __strong__", "**bold** _emphasis_", + "`code` and ``a`b``", "[text](target \"title\")", + "", "[shortcut] [reference][label]", + "[[wiki|text]]", "<https://example.org> <[email protected]>", + "<i>text</i>", "\\*escaped\\*", + "soft\nline", "hard \nline", + "hard\\\nline", "tab\t\nline", + "hard \r\nline", "hard \rline", + "**broken", "[broken](", + "`broken", + }) |source| try check(selected, source); + for (0..128) |byte| { + const char: u8 = @intCast(byte); + const source = [_]u8{ char, 'a', 'b', char, ' ', char, char, 'c', char, char }; + try check(selected, &source); + } +} + +test "syntax plain Markdown keeps block styles without starting the inline parser" { + if (!enabled or !full_grammars) return; + start(std.testing.allocator); + defer stop(); + const source = "# Heading\n\nPlain prose.\n\n indented code\n"; + const styles = try highlightFileRange(std.testing.allocator, "a.md", source, 0, source.len); + defer std.testing.allocator.free(styles); + try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[2]); + try std.testing.expectEqual(@intFromEnum(Syn.none), styles[std.mem.indexOf(u8, source, "Plain").?]); + try std.testing.expectEqual(@intFromEnum(Syn.string), styles[std.mem.indexOf(u8, source, "indented").?]); + for (specs) |spec| { + if (std.mem.eql(u8, spec.name, "markdown_inline")) try std.testing.expect(spec.selected == null); + } +} + +test "syntax highlightDiff colors unified diff lines by prefix" { const diff = "diff --git a/x b/x\n" ++ "--- a/x\n" ++ @@ -850,3 +744,98 @@ test "highlightDiff colors unified diff lines by prefix" { try std.testing.expectEqual(Syn.number, byteSyn(styles, diff, "-old line")); try std.testing.expectEqual(Syn.string, byteSyn(styles, diff, "+new line")); } + +test "syntax result fragments preserve source indentation and inline markup" { + if (!enabled) return; + start(std.testing.allocator); + defer stop(); + const fixtures = [_]struct { path: []const u8, source: []const u8 }{ + .{ .path = "a.zig", .source = " const number = 42; // note" }, + .{ .path = "a.md", .source = "# Heading *slant*" }, + .{ .path = "a.md", .source = "**bold** and `code`" }, + .{ .path = "a.md", .source = " # this is indented code" }, + .{ .path = "a.md", .source = "\t# tab-indented code" }, + .{ .path = "a.py", .source = " return \"hello\"" }, + }; + for (fixtures) |fixture| { + const expected = try highlightFileRange(std.testing.allocator, fixture.path, fixture.source, 0, fixture.source.len); + defer std.testing.allocator.free(expected); + const row = try std.fmt.allocPrint(std.testing.allocator, "{s}:12:3-9 {s}", .{ fixture.path, fixture.source }); + defer std.testing.allocator.free(row); + const actual = try highlightLocations(std.testing.allocator, row, 0, row.len); + defer std.testing.allocator.free(actual); + if (expected.len == 0) { + try std.testing.expectEqual(@as(usize, 0), actual.len); + continue; + } + const code_at = row.len - fixture.source.len; + try std.testing.expectEqualSlices(u8, expected, actual[code_at..]); + for (actual[0..code_at]) |style| try std.testing.expectEqual(@as(u8, 0), style); + } +} + +test "syntax query filtering preserves upstream colors" { + if (!enabled) return; + var allocator: std.heap.DebugAllocator(.{ .stack_trace_frames = 0, .safety = true }) = .init; + defer if (allocator.deinit() != .ok) @panic("leaked syntax query allocations"); + start(allocator.allocator()); + defer stop(); + const source = "// comment\n# Heading *inline*\nconst value = 42;\nif (true) { return \"quoted\"; }\n/* multi\nline */\n"; + inline for (grammar_manifest.all) |grammar| { + if (comptime grammarSelected(grammar)) { + const selected = (try forLang(grammar.name)).?; + const raw_source = @field(ts_queries, grammar.name ++ "_highlights") ++ + (if (comptime std.mem.eql(u8, grammar.name, "typst")) typst_supplement else ""); + var error_offset: u32 = 0; + const raw_query = try ts.Query.create(selected.lang, raw_source, &error_offset); + defer raw_query.destroy(); + var reference = selected; + reference.query = raw_query; + reference.capture_styles = try std.testing.allocator.alloc(u8, raw_query.captureCount()); + defer std.testing.allocator.free(reference.capture_styles); + for (reference.capture_styles, 0..) |*style, id| style.* = @intFromEnum(synFor(raw_query.captureNameForId(@intCast(id)) orelse "")); + const tree = selected.parser.parseString(source, null) orelse return error.ParseFailed; + defer tree.destroy(); + var expected: [source.len]u8 = @splat(0); + var actual: [source.len]u8 = @splat(0); + runQuery(&expected, reference, tree, 0); + runQuery(&actual, selected, tree, 0); + if (!std.mem.eql(u8, &expected, &actual)) std.debug.print("query mismatch: {s}\n", .{grammar.name}); + try std.testing.expectEqualSlices(u8, &expected, &actual); + } + } +} + +test "syntax Zig keyword captures cover both bytes beyond line ten thousand" { + if (!enabled) return; + start(std.testing.allocator); + defer stop(); + const code = "pub fn main() void {\n if (true) return;\n}\n"; + const source = try std.testing.allocator.alloc(u8, 10_001 + code.len); + defer std.testing.allocator.free(source); + @memset(source[0..10_001], '\n'); + @memcpy(source[10_001..], code); + const styles = try highlightFileRange(std.testing.allocator, "a.zig", source, 10_001, source.len); + defer std.testing.allocator.free(styles); + for ([_][]const u8{ "fn", "if" }) |keyword| { + const at = std.mem.indexOf(u8, code, keyword).?; + try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[at]); + try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[at + 1]); + } +} + +test "syntax allocator switching clears default-runtime caches" { + if (!enabled) return; + const source = "fn main() void {}"; + const initial = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len); + std.testing.allocator.free(initial); + start(std.testing.allocator); + const custom = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len); + std.testing.allocator.free(custom); + stop(); + const restored = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len); + defer std.testing.allocator.free(restored); + defer stop(); + try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[0]); + try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[1]); +} |
