summaryrefslogtreecommitdiff
path: root/src/syntax.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/syntax.zig')
-rw-r--r--src/syntax.zig651
1 files changed, 320 insertions, 331 deletions
diff --git a/src/syntax.zig b/src/syntax.zig
index f6e7cae2..77ef80d7 100644
--- a/src/syntax.zig
+++ b/src/syntax.zig
@@ -1,7 +1,3 @@
-//! Tree-sitter syntax highlighting: one style byte per content byte, filled by
-//! running each grammar's highlights.scm query (slurped at build time into the
-//! ts_queries options module). Grammar set is tiered: `zig`; `minimal` (c, cpp,
-//! zig); and `full`, which adds ~24 languages lazily on first use.
const std = @import("std");
const config = @import("pardes_config");
const tracy = @import("tracy.zig");
@@ -16,6 +12,8 @@ const full_grammars = config.syntax_full_grammars;
const ts = if (enabled) @import("tree-sitter") else struct {
pub const Language = opaque {};
pub const Query = opaque {};
+ pub const Parser = opaque {};
+ pub const QueryCursor = opaque {};
};
const ts_queries = if (enabled) @import("ts_queries") else struct {};
@@ -28,7 +26,8 @@ const Spec = struct {
exts: []const []const u8,
language: *const fn () callconv(.c) *const ts.Language,
query_src: []const u8,
- compiled_query: ?*ts.Query = null,
+ selected: ?Selected = null,
+ capture_styles: [256]u8 = undefined,
};
fn grammarSelected(comptime g: grammar_manifest.Grammar) bool {
@@ -46,21 +45,6 @@ fn specCount() comptime_int {
}
return count;
}
-
-// Upstream's typst highlights.scm names its markup honestly —
-// @markup.heading.*, @markup.bold, @markup.italic, @markup.raw.block — and
-// `synFor` now understands that vocabulary, so headings, bold, italics and raw
-// blocks paint on their own. What is left here is exactly the two captures we
-// refuse to map globally:
-//
-// - upstream tags call callees @function/@function.method; mapping "function"
-// in `synFor` would recolour every function call in every language.
-// - upstream tags the code sigil "#" @operator, and mapping "operator" would
-// likewise light up every +, -, == in the codebase.
-//
-// Both are worth colouring *in typst specifically*: the sigil in front of every
-// #let/#if/#import/#call is the visual anchor of the code/markup split, and
-// upstream leaves it uncoloured even though the keyword behind it is not.
const typst_supplement =
\\
\\(call item: (ident) @keyword)
@@ -71,8 +55,69 @@ const typst_supplement =
fn querySrc(comptime g: grammar_manifest.Grammar) []const u8 {
const base = @field(ts_queries, g.name ++ "_highlights");
- if (comptime std.mem.eql(u8, g.name, "typst")) return base ++ typst_supplement;
- return base;
+ const source = if (comptime std.mem.eql(u8, g.name, "typst")) base ++ typst_supplement else base;
+ return comptime colorQuery(source);
+}
+
+fn colorQuery(comptime source: []const u8) []const u8 {
+ @setEvalBranchQuota(2_000_000);
+ var result: [source.len]u8 = undefined;
+ var written: usize = 0;
+ var at: usize = 0;
+ while (at < source.len) {
+ while (at < source.len) {
+ if (std.ascii.isWhitespace(source[at])) {
+ at += 1;
+ } else if (source[at] == ';') {
+ while (at < source.len and source[at] != '\n') at += 1;
+ } else break;
+ }
+ const begin = at;
+ var depth: usize = 0;
+ var colored = false;
+ var complete = false;
+ while (at < source.len) {
+ const byte = source[at];
+ if (complete and (byte == '(' or byte == '[' or byte == '"')) break;
+ if (byte == ';') {
+ while (at < source.len and source[at] != '\n') at += 1;
+ continue;
+ }
+ if (byte == '"') {
+ at += 1;
+ while (at < source.len) : (at += 1) {
+ if (source[at] == '\\') {
+ at += 1;
+ } else if (source[at] == '"') {
+ at += 1;
+ break;
+ }
+ }
+ if (depth == 0) complete = true;
+ continue;
+ }
+ if (byte == '(' or byte == '[') depth += 1;
+ if (byte == ')' or byte == ']') {
+ depth -= 1;
+ if (depth == 0) complete = true;
+ }
+ if (byte == '@') {
+ const name = at + 1;
+ at = name;
+ while (at < source.len and (std.ascii.isAlphanumeric(source[at]) or
+ source[at] == '_' or source[at] == '.' or source[at] == '-')) at += 1;
+ colored = colored or synFor(source[name..at]) != .none;
+ continue;
+ }
+ at += 1;
+ }
+ if (colored) {
+ @memcpy(result[written..][0 .. at - begin], source[begin..at]);
+ written += at - begin;
+ }
+ }
+ const filtered = result[0..written].*;
+ return &filtered;
}
fn initSpecs() [specCount()]Spec {
@@ -151,6 +196,7 @@ fn syntaxFree(ptr: ?*anyopaque) callconv(.c) void {
pub fn start(gpa: std.mem.Allocator) void {
if (comptime enabled) {
std.debug.assert(!syntax_started);
+ stop();
syntax_allocator = gpa;
syntax_started = true;
ts_set_allocator(syntaxAlloc, syntaxCalloc, syntaxRealloc, syntaxFree);
@@ -159,10 +205,13 @@ pub fn start(gpa: std.mem.Allocator) void {
pub fn stop() void {
if (comptime enabled) {
- std.debug.assert(syntax_started);
for (&specs) |*spec| {
- if (spec.compiled_query) |query| query.destroy();
- spec.compiled_query = null;
+ if (spec.selected) |selected| {
+ selected.cursor.destroy();
+ selected.parser.destroy();
+ selected.query.destroy();
+ }
+ spec.selected = null;
}
ts_set_allocator(null, null, null, null);
syntax_started = false;
@@ -170,17 +219,41 @@ pub fn stop() void {
}
}
-const Selected = struct { name: []const u8, lang: *const ts.Language, query: *ts.Query };
+const Selected = struct {
+ name: []const u8,
+ lang: *const ts.Language,
+ query: *ts.Query,
+ parser: *ts.Parser,
+ cursor: *ts.QueryCursor,
+ capture_styles: []u8,
+};
-// NOTE: don't lang.destroy() — the tree_sitter_*() languages are static
-// singletons reused on every open; destroying one use-after-frees the next.
fn ensure(spec: *Spec) !Selected {
+ if (spec.selected) |selected| return selected;
const lang = spec.language();
- if (spec.compiled_query) |query| return .{ .name = spec.name, .lang = lang, .query = query };
var error_offset: u32 = 0;
const query = try ts.Query.create(lang, spec.query_src, &error_offset);
- spec.compiled_query = query;
- return .{ .name = spec.name, .lang = lang, .query = query };
+ errdefer query.destroy();
+ if (query.captureCount() > spec.capture_styles.len) return error.TooManyCaptures;
+ const capture_styles = spec.capture_styles[0..query.captureCount()];
+ for (capture_styles, 0..) |*style, id| {
+ const name = query.captureNameForId(@intCast(id)) orelse "";
+ style.* = @intFromEnum(synFor(name));
+ if (style.* == 0) query.disableCapture(name);
+ }
+ const parser = ts.Parser.create();
+ errdefer parser.destroy();
+ try parser.setLanguage(lang);
+ const selected: Selected = .{
+ .name = spec.name,
+ .lang = lang,
+ .query = query,
+ .parser = parser,
+ .cursor = ts.QueryCursor.create(),
+ .capture_styles = capture_styles,
+ };
+ spec.selected = selected;
+ return selected;
}
fn forExt(ext: []const u8) !?Selected {
@@ -226,30 +299,20 @@ fn forLang(name: []const u8) !?Selected {
}
return null;
}
-
-/// The caller owns the cursor: `ts_query_cursor_exec` fully resets its state,
-/// so one cursor serves any number of trees, and the per-row pass would
-/// otherwise create and destroy one — three allocations against the shared
-/// tree-sitter arena — for every row of a results buffer.
-fn runQuery(styles: []u8, sel: Selected, tree: *ts.Tree, base: usize, cursor: *ts.QueryCursor) void {
+fn runQuery(styles: []u8, sel: Selected, tree: *ts.Tree, base: usize) void {
+ const cursor = sel.cursor;
cursor.exec(sel.query, tree.rootNode());
while (cursor.nextMatch()) |match| {
for (match.captures) |cap| {
- const syn = synFor(sel.query.captureNameForId(cap.index) orelse "");
- if (syn == .none) continue;
+ const style = sel.capture_styles[cap.index];
+ if (style == 0) continue;
const b = @min(base + cap.node.startByte(), styles.len);
const end = @min(base + @as(usize, cap.node.endByte()), styles.len);
- if (end > b) @memset(styles[b..end], @intFromEnum(syn));
+ if (end > b) @memset(styles[b..end], style);
}
}
}
-// Queries compile on first use (`ensure`), never at startup. Pre-compiling the
-// compact tier in Pardes.init cost EVERY boot ~120ms of ts_query__perform_analysis
-// (55% of a Debug startup) to save ~40ms on the first .zig/.c/.cpp open — a pane
-// of prose or a shell paid for a language it never opened. Grammar availability
-// is unchanged; only the timing moved.
-
fn synFor(name: []const u8) Syn {
for ([_]struct { []const u8, Syn }{
.{ "comment", .comment },
@@ -266,19 +329,6 @@ fn synFor(name: []const u8) Syn {
.{ "title", .keyword },
.{ "uri", .string },
.{ "reference", .number },
-
- // Markup grammars (markdown, typst) name prose constructs in their own
- // vocabulary rather than the code vocabulary above, so none of the
- // needles so far reach them. These are appended, and first-match-wins
- // makes that strictly additive; the needles below were audited across
- // all 27 shipped queries and occur only in the markdown and typst ones,
- // so no other language is recoloured.
- //
- // The slot assignment is forced by the palette being four wide and by
- // `synStyle` attaching the real BOLD attribute to exactly two of them,
- // `keyword` and `comment`: headings take `keyword`, so bold spans have
- // to land on `comment` to render actually bold. Nothing is left that
- // renders italic, so emphasis can only get a colour shift (`number`).
.{ "heading", .keyword },
.{ "strong", .comment },
.{ "bold", .comment },
@@ -291,75 +341,31 @@ fn synFor(name: []const u8) Syn {
}
return .none;
}
-
-/// One Syn byte per content byte in [start, end). Caller frees.
-pub fn highlightFileRange(gpa: std.mem.Allocator, path: []const u8, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 {
- const tz = tracy.zone(@src(), "highlightFileRange");
- defer tz.end();
+pub fn highlightFileRange(gpa: std.mem.Allocator, path: []const u8, content: []const u8, start_raw: usize, end_raw: usize) ![]u8 {
+ const zone = tracy.zone(@src(), "highlightFileRange");
+ defer zone.end();
if (!enabled) return &.{};
- const ext = std.fs.path.extension(path);
- const selected = (forExt(ext) catch return &.{}) orelse return &.{};
-
- const start_byte = @min(start_byte_raw, content.len);
- const end_byte = @max(start_byte, @min(end_byte_raw, content.len));
+ const selected = (try forExt(std.fs.path.extension(path))) orelse return &.{};
+ const start_byte = @min(start_raw, content.len);
+ const end_byte = @max(start_byte, @min(end_raw, content.len));
const source = content[start_byte..end_byte];
const styles = try gpa.alloc(u8, source.len);
- errdefer gpa.free(styles);
@memset(styles, 0);
-
- const parser = ts.Parser.create();
- defer parser.destroy();
- parser.setLanguage(selected.lang) catch return styles;
- const cursor = ts.QueryCursor.create();
- defer cursor.destroy();
- paintWith(styles, source, selected, parser, cursor, true);
+ paint(styles, source, selected);
return styles;
}
-/// Parse `source` and write its style bytes into `styles`, with a parser and a
-/// query cursor the caller owns, so the per-row pass below can run a whole
-/// results buffer through one of each.
-///
-/// `inject` is off for a single row. Both injection passes build a SECOND
-/// parser of their own — per fenced block, per `inline` node — which is
-/// amortised over a document and absurd over one truncated grep row that
-/// almost never contains a fenced block to begin with.
-fn paintWith(
- styles: []u8,
- source: []const u8,
- selected: Selected,
- parser: *ts.Parser,
- cursor: *ts.QueryCursor,
- inject: bool,
-) void {
- const tree = parser.parseString(source, null) orelse return;
+fn paint(styles: []u8, source: []const u8, selected: Selected) void {
+ const tree = selected.parser.parseString(source, null) orelse return;
defer tree.destroy();
-
- runQuery(styles, selected, tree, 0, cursor);
- if (!inject) return;
-
- if (InjectSite.forGrammar(selected.name)) |site| injectCodeBlocks(styles, source, tree.rootNode(), site);
- // Disjoint from the fenced-block pass above: `code_fence_content` is never
- // an `inline` node, so the two never write the same byte.
- if (std.mem.eql(u8, selected.name, "markdown")) injectMarkdownInline(styles, source, tree.rootNode());
+ runQuery(styles, selected, tree, 0);
+ if (std.mem.eql(u8, selected.name, "markdown")) {
+ inject(styles, source, tree.rootNode(), true);
+ } else if (std.mem.eql(u8, selected.name, "typst")) {
+ inject(styles, source, tree.rootNode(), false);
+ }
}
-/// A results buffer — every search, grep and language answer in this program —
-/// coloured as the CODE it is quoting.
-///
-/// The rows look like `src/look.zig:718:12-16 fn grepText(path: []const u8...`:
-/// a location, a space, and a piece of some file. The location names the file,
-/// the file names the grammar, and the rest of the row is a fragment of that
-/// language — so a +Grep over Zig reads as Zig and one over Markdown does not
-/// pretend to. `look.parsePathLine` decides what counts as a location, which is
-/// the same primitive n/N walks these buffers with, so the two agree by
-/// construction about which rows are locations.
-///
-/// ONE PARSER AND ONE CURSOR for the whole buffer, and the buffer is coloured
-/// once when it is filled rather than per visible window (file_pane
-/// `refreshHighlights`) — the rows are independent, so a window pass buys no
-/// fidelity and pays a burst of parses, cursors and first-time query compiles
-/// on every scroll that outran the covered range.
pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 {
const tz = tracy.zone(@src(), "highlightLocations");
defer tz.end();
@@ -370,19 +376,6 @@ pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byt
const styles = try gpa.alloc(u8, source.len);
errdefer gpa.free(styles);
@memset(styles, 0);
-
- // Both are created on the first row that needs them and kept for the rest;
- // `held` is the language the parser is currently set to.
- var parser: ?*ts.Parser = null;
- defer if (parser) |ptr| ptr.destroy();
- var cursor: ?*ts.QueryCursor = null;
- defer if (cursor) |ptr| ptr.destroy();
- var held: ?Selected = null;
- // Consecutive rows of a results buffer are overwhelmingly the same file,
- // and `forExt` is a linear walk of 29 specs and their extension lists. One
- // remembered answer collapses that to a string compare — including for the
- // rows that match NOTHING (a jumplist `@p3:10`, a `.lock`, a `.txt`),
- // which otherwise pay the whole failing scan every time.
var memo_ext: []const u8 = "\x00";
var memo: ?Selected = null;
var painted = false;
@@ -398,100 +391,61 @@ pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byt
memo = forExt(ext) catch null;
}
const selected = memo orelse continue;
- if (parser == null) parser = ts.Parser.create();
- if (cursor == null) cursor = ts.QueryCursor.create();
- if (held == null or held.?.lang != selected.lang) {
- // `held` is cleared FIRST: a failed `setLanguage` has already set
- // the parser's language to null, so leaving `held` on the previous
- // grammar makes every later row of it skip the call and parse
- // against nothing — the rest of the buffer silently loses colour.
- held = null;
- parser.?.setLanguage(selected.lang) catch continue;
- held = selected;
- }
- paintWith(styles[offset + code.at ..][0..code.text.len], code.text, selected, parser.?, cursor.?, false);
+ paint(styles[offset + code.at ..][0..code.text.len], code.text, selected);
painted = true;
}
- // NOTHING TO PAINT IS NOTHING TO KEEP. `recolorSyntax` skips a pane whose
- // highlights are empty, and every output buffer without locations in it —
- // +Help, +Config, +Messages, +Errors — would otherwise hand the renderer a
- // full-length run of zeroes and make it walk every visible grapheme, every
- // frame, to paint nothing.
if (!painted) {
gpa.free(styles);
return &.{};
}
return styles;
}
-
-/// The `<path> <code>` split of one results row, or null when the row is not
-/// one. A row qualifies when its FIRST whitespace-delimited token is entirely a
-/// look target — the whole token, so `see:` in prose does not count — and
-/// something follows it.
fn codeAfterLocation(line: []const u8) ?struct { path: []const u8, at: usize, text: []const u8 } {
const token_end = std.mem.indexOfAny(u8, line, " \t") orelse return null;
if (token_end == 0) return null;
const token = line[0..token_end];
const target = look.parsePathLine(token);
if (target.end != token.len) return null;
- // A bare word is not a location: `main.zig` alone is a filename, but a
- // results row is `main.zig:12:3`, and without that a prose line whose
- // first word happens to end in `.md` would colour the rest of a sentence.
if (target.at.line == 0) return null;
- var at = token_end;
- while (at < line.len and (line[at] == ' ' or line[at] == '\t')) at += 1;
+ const at = token_end + 1;
if (at >= line.len) return null;
return .{ .path = target.path, .at = at, .text = line[at..] };
}
-
-// Markdown fenced blocks and Typst raw blocks are the same construct — a
-// language tag plus a literal payload — under different node shapes, so one
-// walker drives both and only the (lang, content) extraction differs.
-const InjectSite = enum {
- markdown_fence,
- typst_raw,
-
- fn forGrammar(name: []const u8) ?InjectSite {
- if (std.mem.eql(u8, name, "markdown")) return .markdown_fence;
- if (std.mem.eql(u8, name, "typst")) return .typst_raw;
- return null;
- }
-
- fn blockKind(self: InjectSite) []const u8 {
- return switch (self) {
- .markdown_fence => "fenced_code_block",
- .typst_raw => "raw_blck",
- };
- }
-
- /// null when the block carries no language tag (an untagged fence, or a
- /// Typst raw block written without one) — nothing to inject, leave it alone.
- fn parts(self: InjectSite, block: ts.Node) ?struct { lang: ts.Node, content: ts.Node } {
- switch (self) {
- .markdown_fence => {
- const info = childOfKind(block, "info_string") orelse return null;
- return .{
- .lang = childOfKind(info, "language") orelse return null,
- .content = childOfKind(block, "code_fence_content") orelse return null,
- };
- },
- .typst_raw => return .{
- .lang = block.childByFieldName("lang") orelse return null,
- .content = childOfKind(block, "blob") orelse return null,
- },
- }
+fn inject(styles: []u8, source: []const u8, node: ts.Node, markdown: bool) void {
+ const kind = node.kind();
+ if (markdown and std.mem.eql(u8, kind, "inline")) {
+ const begin: usize = node.startByte();
+ const end: usize = node.endByte();
+ if (begin >= end or end > source.len) return;
+ const text = source[begin..end];
+ // Every colored inline capture requires one of these delimiters.
+ if (std.mem.indexOfAny(u8, text, "*_`[<\\\r\n") == null) return;
+ const selected = (forLang("markdown_inline") catch return) orelse return;
+ const tree = selected.parser.parseString(text, null) orelse return;
+ defer tree.destroy();
+ runQuery(styles, selected, tree, begin);
+ return;
}
-};
-
-fn injectCodeBlocks(styles: []u8, source: []const u8, node: ts.Node, site: InjectSite) void {
- if (std.mem.eql(u8, node.kind(), site.blockKind())) {
- highlightCodeBlock(styles, source, node, site);
+ if (std.mem.eql(u8, kind, if (markdown) "fenced_code_block" else "raw_blck")) {
+ const lang = if (markdown) blk: {
+ const info = childOfKind(node, "info_string") orelse return;
+ break :blk childOfKind(info, "language") orelse return;
+ } else node.childByFieldName("lang") orelse return;
+ const content = childOfKind(node, if (markdown) "code_fence_content" else "blob") orelse return;
+ const selected = (forLang(source[lang.startByte()..lang.endByte()]) catch return) orelse return;
+ const begin: usize = content.startByte();
+ const end: usize = content.endByte();
+ if (begin > end or end > source.len) return;
+ const tree = selected.parser.parseString(source[begin..end], null) orelse return;
+ defer tree.destroy();
+ @memset(styles[begin..end], 0);
+ runQuery(styles, selected, tree, begin);
return;
}
var i: u32 = 0;
const count = node.childCount();
while (i < count) : (i += 1) {
- if (node.child(i)) |c| injectCodeBlocks(styles, source, c, site);
+ if (node.child(i)) |child| inject(styles, source, child, markdown);
}
}
@@ -506,79 +460,6 @@ fn childOfKind(node: ts.Node, kind: []const u8) ?ts.Node {
return null;
}
-fn highlightCodeBlock(styles: []u8, source: []const u8, block: ts.Node, site: InjectSite) void {
- const p = site.parts(block) orelse return;
- const langtext = source[p.lang.startByte()..p.lang.endByte()];
- const sub_sel = (forLang(langtext) catch return) orelse return;
- const cs: usize = p.content.startByte();
- const ce: usize = p.content.endByte();
- if (ce > source.len or cs > ce) return;
-
- const parser = ts.Parser.create();
- defer parser.destroy();
- parser.setLanguage(sub_sel.lang) catch return;
- const tree = parser.parseString(source[cs..ce], null) orelse return;
- defer tree.destroy();
- // The outer grammar already painted these bytes (Typst blankets the whole
- // raw block `string`), and runQuery only writes bytes it captures, so the
- // outer colour would survive as a wash behind the injected code. The sub
- // grammar owns the payload outright: clear it first.
- @memset(styles[cs..ce], @intFromEnum(Syn.none));
- const cursor = ts.QueryCursor.create();
- defer cursor.destroy();
- runQuery(styles, sub_sel, tree, cs, cursor);
-}
-
-// tree-sitter-markdown is a split grammar: the block parser bottoms out at named
-// `inline` nodes whose bytes it never looks inside, and emphasis /
-// strong_emphasis / code_span exist only in the companion inline parser. So
-// every `inline` node is re-parsed with `markdown_inline`, which is what
-// upstream's injections.scm, helix and nvim all do (per node, uncombined — the
-// stray block_continuation markers inside a multi-line paragraph's range are
-// just plain text to the inline parser).
-//
-// The grammar is resolved and the parser built once per file, not once per node:
-// only the parse is inherently per node.
-fn injectMarkdownInline(styles: []u8, source: []const u8, root: ts.Node) void {
- // Absent in builds below the `full` tier — nothing to inject, leave the
- // block grammar's colours alone.
- const sub_sel = (forLang("markdown_inline") catch return) orelse return;
- const parser = ts.Parser.create();
- defer parser.destroy();
- parser.setLanguage(sub_sel.lang) catch return;
- inlineNodes(styles, source, root, sub_sel, parser);
-}
-
-fn inlineNodes(styles: []u8, source: []const u8, node: ts.Node, sel: Selected, parser: *ts.Parser) void {
- if (std.mem.eql(u8, node.kind(), "inline")) {
- highlightInline(styles, source, node, sel, parser);
- return; // `inline` nodes never nest
- }
- var i: u32 = 0;
- const count = node.childCount();
- while (i < count) : (i += 1) {
- if (node.child(i)) |c| inlineNodes(styles, source, c, sel, parser);
- }
-}
-
-fn highlightInline(styles: []u8, source: []const u8, node: ts.Node, sel: Selected, parser: *ts.Parser) void {
- const s: usize = node.startByte();
- const e: usize = node.endByte();
- if (s >= e or e > source.len) return; // an empty atx heading has a zero-length inline
- const tree = parser.parseString(source[s..e], null) orelse return;
- defer tree.destroy();
- // Deliberately NOT clearing the range first, unlike highlightCodeBlock: the
- // block query already painted a heading's inline text `keyword`, and that is
- // the colour the heading must keep wherever the inline pass captures
- // nothing. Painting over instead of resetting is what makes *italic* inside
- // a heading recolour while the rest of the heading stays heading-coloured.
- const cursor = ts.QueryCursor.create();
- defer cursor.destroy();
- runQuery(styles, sel, tree, s, cursor);
-}
-
-/// One Syn byte per byte of content[start, end) for unified diffs/patches.
-/// Pure byte scan; independent of tree-sitter and the `enabled` flag. Caller frees.
pub fn highlightDiff(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 {
const start_byte = @min(start_byte_raw, content.len);
const end_byte = @max(start_byte, @min(end_byte_raw, content.len));
@@ -610,14 +491,11 @@ fn diffLineSyn(line: []const u8) Syn {
return .none;
}
-test "a results row is coloured by the file its location names" {
+test "syntax a results row is coloured by the file its location names" {
if (!enabled) return;
start(std.testing.allocator);
defer stop();
const gpa = std.testing.allocator;
-
- // Two rows quoting two languages, plus a row that is not a location and a
- // location with nothing after it.
const content =
"src/a.zig:1:1 const S = struct {};\n" ++
"src/b.md:2:1 # heading\n" ++
@@ -626,53 +504,31 @@ test "a results row is coloured by the file its location names" {
const styles = try highlightLocations(gpa, content, 0, content.len);
defer gpa.free(styles);
try std.testing.expectEqual(content.len, styles.len);
-
- // The LOCATION itself is left alone — it is not code, and colouring it as
- // code is how a path starts looking like a keyword.
for (styles[0.."src/a.zig:1:1".len]) |b| try std.testing.expectEqual(@as(u8, 0), b);
-
- // ...and `struct` in the Zig row is a keyword, which is only true if the
- // grammar was chosen from `a.zig` rather than from the buffer's own name.
- //
- // `struct` and not `const`: tree-sitter-zig captures `const` as
- // `@type.qualifier`, which `synFor` maps to nothing — a real property of
- // the shipped query rather than of this pass, and the reason the first
- // version of this test failed.
const zig_kw = std.mem.indexOf(u8, content, "struct").?;
try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw]);
try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw + 5]);
-
- // A row with no location contributes nothing...
const prose = std.mem.indexOf(u8, content, "just some prose").?;
for (styles[prose .. prose + 14]) |b| try std.testing.expectEqual(@as(u8, 0), b);
- // ...and neither does a location with no code after it.
const bare = std.mem.indexOf(u8, content, "src/c.zig").?;
for (styles[bare..]) |b| try std.testing.expectEqual(@as(u8, 0), b);
}
-test "a buffer with no locations in it keeps no highlights at all" {
+test "syntax a buffer with no locations in it keeps no highlights at all" {
if (!enabled) return;
start(std.testing.allocator);
defer stop();
- // +Help, +Config, +Messages: prose. An all-zero run of styles is not the
- // same as none — `recolorSyntax` skips a pane whose highlights are EMPTY,
- // and returning a full-length run of zeroes made it walk every visible
- // grapheme every frame to paint nothing.
const content = "nothing has been said yet\n0: save: AccessDenied (x2)\n";
const styles = try highlightLocations(std.testing.allocator, content, 0, content.len);
defer std.testing.allocator.free(styles);
try std.testing.expectEqual(@as(usize, 0), styles.len);
}
-test "codeAfterLocation takes whole-token locations and nothing else" {
- // A grep row: path, line, column range, then the quoted source.
+test "syntax codeAfterLocation takes whole-token locations and nothing else" {
const got = codeAfterLocation("src/x.zig:7:2-9 fn main() void {") orelse
return error.ShouldBeALocation;
try std.testing.expectEqualStrings("src/x.zig", got.path);
try std.testing.expectEqualStrings("fn main() void {", got.text);
-
- // Not locations: a bare filename (no line), prose with a colon, a token
- // that only PARTLY parses, and a location with nothing after it.
try std.testing.expect(codeAfterLocation("main.zig some words") == null);
try std.testing.expect(codeAfterLocation("note: this is prose") == null);
try std.testing.expect(codeAfterLocation("src/x.zig:7:2x rest") == null);
@@ -681,7 +537,7 @@ test "codeAfterLocation takes whole-token locations and nothing else" {
try std.testing.expect(codeAfterLocation(" leading space") == null);
}
-test "tree-sitter allocator callbacks preserve and free exact allocations" {
+test "syntax tree-sitter allocator callbacks preserve and free exact allocations" {
syntax_allocator = std.testing.allocator;
defer syntax_allocator = undefined;
@@ -700,7 +556,7 @@ test "tree-sitter allocator callbacks preserve and free exact allocations" {
live = null;
}
-test "default full grammar set highlights Typst source" {
+test "syntax default full grammar set highlights Typst source" {
if (!enabled or !full_grammars) return;
start(std.testing.allocator);
defer stop();
@@ -723,7 +579,7 @@ test "default full grammar set highlights Typst source" {
try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(short_ext[keyword_at])));
}
-test "Typst markup constructs paint and raw blocks inject their language" {
+test "syntax Typst markup constructs paint and raw blocks inject their language" {
if (!enabled or !full_grammars) return;
start(std.testing.allocator);
defer stop();
@@ -748,25 +604,14 @@ test "Typst markup constructs paint and raw blocks inject their language" {
return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]);
}
}.f;
-
- // marker and text of the heading both take the bold accent
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "= Heading", 0));
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Heading", 0));
- // *bold* is @markup.bold, which lands on the one slot that still renders
- // with the real bold attribute now that headings own `keyword`.
try std.testing.expectEqual(Syn.comment, synAt(styles, source, "bold", 0));
try std.testing.expectEqual(Syn.string, synAt(styles, source, "raw` inline", 0));
- // the callee and the code sigil in front of it
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "emit", 0));
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "#emit", 0));
-
- // fence and lang tag keep the literal colour of the raw block...
try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0));
try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 3));
- // ...while the blob is reset and re-painted by the injected zig grammar. Its
- // `fn` and `99` prove the injection ran; `widget` proves the reset, since the
- // zig query names it @function (Syn.none here) and without clearing the blob
- // first the raw block's `string` would still be washing over it.
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn widget", 0));
try std.testing.expectEqual(Syn.number, synAt(styles, source, "99", 0));
try std.testing.expectEqual(Syn.none, synAt(styles, source, "widget", 0));
@@ -778,7 +623,7 @@ test "Typst markup constructs paint and raw blocks inject their language" {
try std.testing.expectEqual(Syn.string, synAt(styles, source, "\"arg\"", 0));
}
-test "markdown highlights markup, injects inline spans and fenced code blocks" {
+test "syntax markdown highlights markup, injects inline spans and fenced code blocks" {
if (!enabled or !full_grammars) return;
start(std.testing.allocator);
defer stop();
@@ -799,31 +644,80 @@ test "markdown highlights markup, injects inline spans and fenced code blocks" {
return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]);
}
}.f;
-
- // The block grammar washes the whole heading `keyword`; the inline pass then
- // paints the emphasis over it without resetting, so the heading keeps its
- // colour everywhere the emphasis is not.
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Title", 0));
try std.testing.expectEqual(Syn.number, synAt(styles, source, "slant", 0));
-
- // bold/italic/code-span live only in the inline grammar, so all three prove
- // the inline injection ran.
try std.testing.expectEqual(Syn.comment, synAt(styles, source, "stout", 0));
try std.testing.expectEqual(Syn.number, synAt(styles, source, "lean", 0));
try std.testing.expectEqual(Syn.string, synAt(styles, source, "snippet", 0));
-
- // the fence is inside the block query's @text.literal wash...
try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0));
- // ...while the payload is cleared and re-painted by the injected zig
- // grammar: `fn` and `77` prove the injection ran, and `gadget` proves the
- // clear, since the zig query names it @function (Syn.none here) and without
- // clearing first the fence's `string` would still be washing over it.
try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn gadget", 0));
try std.testing.expectEqual(Syn.number, synAt(styles, source, "77", 0));
try std.testing.expectEqual(Syn.none, synAt(styles, source, "gadget", 0));
}
-test "highlightDiff colors unified diff lines by prefix" {
+test "syntax inline fast path agrees with full Markdown query" {
+ if (!enabled or !full_grammars) return;
+ start(std.testing.allocator);
+ defer stop();
+ const selected = (try forLang("markdown_inline")).?;
+ const check = struct {
+ fn compare(sel: Selected, source: []const u8) !void {
+ const tree = sel.parser.parseString(source, null) orelse return error.ParseFailed;
+ defer tree.destroy();
+ const expected = try std.testing.allocator.alloc(u8, source.len);
+ defer std.testing.allocator.free(expected);
+ const actual = try std.testing.allocator.alloc(u8, source.len);
+ defer std.testing.allocator.free(actual);
+ for ([_]Syn{ .none, .keyword }) |background| {
+ @memset(expected, @intFromEnum(background));
+ @memset(actual, @intFromEnum(background));
+ runQuery(expected, sel, tree, 0);
+ inject(actual, source, tree.rootNode(), true);
+ if (!std.mem.eql(u8, expected, actual)) std.debug.print("inline mismatch: {s}\n", .{source});
+ try std.testing.expectEqualSlices(u8, expected, actual);
+ }
+ }
+ }.compare;
+ for ([_][]const u8{
+ "", "plain prose",
+ "ação Ελληνικά 日本語 🙂",
+ "123 456", "tabs\tand spaces",
+ "'quoted' (parentheses) \"double quotes\"", "https://example.org [email protected]",
+ "&amp; &#10; &#x20;", "~~struck~~ $formula$",
+ "*emphasis* __strong__", "**bold** _emphasis_",
+ "`code` and ``a`b``", "[text](target \"title\")",
+ "![description](image)", "[shortcut] [reference][label]",
+ "[[wiki|text]]", "<https://example.org> <[email protected]>",
+ "<i>text</i>", "\\*escaped\\*",
+ "soft\nline", "hard \nline",
+ "hard\\\nline", "tab\t\nline",
+ "hard \r\nline", "hard \rline",
+ "**broken", "[broken](",
+ "`broken",
+ }) |source| try check(selected, source);
+ for (0..128) |byte| {
+ const char: u8 = @intCast(byte);
+ const source = [_]u8{ char, 'a', 'b', char, ' ', char, char, 'c', char, char };
+ try check(selected, &source);
+ }
+}
+
+test "syntax plain Markdown keeps block styles without starting the inline parser" {
+ if (!enabled or !full_grammars) return;
+ start(std.testing.allocator);
+ defer stop();
+ const source = "# Heading\n\nPlain prose.\n\n indented code\n";
+ const styles = try highlightFileRange(std.testing.allocator, "a.md", source, 0, source.len);
+ defer std.testing.allocator.free(styles);
+ try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[2]);
+ try std.testing.expectEqual(@intFromEnum(Syn.none), styles[std.mem.indexOf(u8, source, "Plain").?]);
+ try std.testing.expectEqual(@intFromEnum(Syn.string), styles[std.mem.indexOf(u8, source, "indented").?]);
+ for (specs) |spec| {
+ if (std.mem.eql(u8, spec.name, "markdown_inline")) try std.testing.expect(spec.selected == null);
+ }
+}
+
+test "syntax highlightDiff colors unified diff lines by prefix" {
const diff =
"diff --git a/x b/x\n" ++
"--- a/x\n" ++
@@ -850,3 +744,98 @@ test "highlightDiff colors unified diff lines by prefix" {
try std.testing.expectEqual(Syn.number, byteSyn(styles, diff, "-old line"));
try std.testing.expectEqual(Syn.string, byteSyn(styles, diff, "+new line"));
}
+
+test "syntax result fragments preserve source indentation and inline markup" {
+ if (!enabled) return;
+ start(std.testing.allocator);
+ defer stop();
+ const fixtures = [_]struct { path: []const u8, source: []const u8 }{
+ .{ .path = "a.zig", .source = " const number = 42; // note" },
+ .{ .path = "a.md", .source = "# Heading *slant*" },
+ .{ .path = "a.md", .source = "**bold** and `code`" },
+ .{ .path = "a.md", .source = " # this is indented code" },
+ .{ .path = "a.md", .source = "\t# tab-indented code" },
+ .{ .path = "a.py", .source = " return \"hello\"" },
+ };
+ for (fixtures) |fixture| {
+ const expected = try highlightFileRange(std.testing.allocator, fixture.path, fixture.source, 0, fixture.source.len);
+ defer std.testing.allocator.free(expected);
+ const row = try std.fmt.allocPrint(std.testing.allocator, "{s}:12:3-9 {s}", .{ fixture.path, fixture.source });
+ defer std.testing.allocator.free(row);
+ const actual = try highlightLocations(std.testing.allocator, row, 0, row.len);
+ defer std.testing.allocator.free(actual);
+ if (expected.len == 0) {
+ try std.testing.expectEqual(@as(usize, 0), actual.len);
+ continue;
+ }
+ const code_at = row.len - fixture.source.len;
+ try std.testing.expectEqualSlices(u8, expected, actual[code_at..]);
+ for (actual[0..code_at]) |style| try std.testing.expectEqual(@as(u8, 0), style);
+ }
+}
+
+test "syntax query filtering preserves upstream colors" {
+ if (!enabled) return;
+ var allocator: std.heap.DebugAllocator(.{ .stack_trace_frames = 0, .safety = true }) = .init;
+ defer if (allocator.deinit() != .ok) @panic("leaked syntax query allocations");
+ start(allocator.allocator());
+ defer stop();
+ const source = "// comment\n# Heading *inline*\nconst value = 42;\nif (true) { return \"quoted\"; }\n/* multi\nline */\n";
+ inline for (grammar_manifest.all) |grammar| {
+ if (comptime grammarSelected(grammar)) {
+ const selected = (try forLang(grammar.name)).?;
+ const raw_source = @field(ts_queries, grammar.name ++ "_highlights") ++
+ (if (comptime std.mem.eql(u8, grammar.name, "typst")) typst_supplement else "");
+ var error_offset: u32 = 0;
+ const raw_query = try ts.Query.create(selected.lang, raw_source, &error_offset);
+ defer raw_query.destroy();
+ var reference = selected;
+ reference.query = raw_query;
+ reference.capture_styles = try std.testing.allocator.alloc(u8, raw_query.captureCount());
+ defer std.testing.allocator.free(reference.capture_styles);
+ for (reference.capture_styles, 0..) |*style, id| style.* = @intFromEnum(synFor(raw_query.captureNameForId(@intCast(id)) orelse ""));
+ const tree = selected.parser.parseString(source, null) orelse return error.ParseFailed;
+ defer tree.destroy();
+ var expected: [source.len]u8 = @splat(0);
+ var actual: [source.len]u8 = @splat(0);
+ runQuery(&expected, reference, tree, 0);
+ runQuery(&actual, selected, tree, 0);
+ if (!std.mem.eql(u8, &expected, &actual)) std.debug.print("query mismatch: {s}\n", .{grammar.name});
+ try std.testing.expectEqualSlices(u8, &expected, &actual);
+ }
+ }
+}
+
+test "syntax Zig keyword captures cover both bytes beyond line ten thousand" {
+ if (!enabled) return;
+ start(std.testing.allocator);
+ defer stop();
+ const code = "pub fn main() void {\n if (true) return;\n}\n";
+ const source = try std.testing.allocator.alloc(u8, 10_001 + code.len);
+ defer std.testing.allocator.free(source);
+ @memset(source[0..10_001], '\n');
+ @memcpy(source[10_001..], code);
+ const styles = try highlightFileRange(std.testing.allocator, "a.zig", source, 10_001, source.len);
+ defer std.testing.allocator.free(styles);
+ for ([_][]const u8{ "fn", "if" }) |keyword| {
+ const at = std.mem.indexOf(u8, code, keyword).?;
+ try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[at]);
+ try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[at + 1]);
+ }
+}
+
+test "syntax allocator switching clears default-runtime caches" {
+ if (!enabled) return;
+ const source = "fn main() void {}";
+ const initial = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len);
+ std.testing.allocator.free(initial);
+ start(std.testing.allocator);
+ const custom = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len);
+ std.testing.allocator.free(custom);
+ stop();
+ const restored = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len);
+ defer std.testing.allocator.free(restored);
+ defer stop();
+ try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[0]);
+ try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[1]);
+}