const std = @import("std"); const config = @import("pardes_config"); const tracy = @import("tracy.zig"); const grammar_manifest = @import("grammar_manifest.zig"); const look = @import("look.zig"); const regexp = @import("regexp.zig"); const c_heap = @import("c_heap"); const diff = @import("diff.zig"); test { _ = diff; } pub const enabled = config.syntax_highlighting; const zig_grammar = config.syntax_zig_grammar; const minimal_grammars = config.syntax_minimal_grammars; const full_grammars = config.syntax_full_grammars; const ts = if (enabled) @import("tree-sitter") else struct { pub const Language = opaque {}; pub const Query = opaque {}; pub const Parser = opaque {}; pub const QueryCursor = opaque {}; }; const ts_queries = if (enabled) @import("ts_queries") else struct {}; const LanguageFn = *const fn () callconv(.c) *const ts.Language; pub const Syn = enum(u8) { none, keyword, string, number, comment }; const Spec = struct { name: []const u8, exts: []const []const u8, language: *const fn () callconv(.c) *const ts.Language, query_src: []const u8, /// helix's textobjects.scm for the grammar (vendor/queries), "" for none objects_src: []const u8 = "", objects: ?*ts.Query = null, objects_failed: bool = false, selected: ?Selected = null, capture_styles: [256]u8 = undefined, }; fn grammarSelected(comptime g: grammar_manifest.Grammar) bool { return switch (g.tier) { .zig => zig_grammar, .minimal => minimal_grammars, .full => full_grammars, }; } fn specCount() comptime_int { var count = 0; for (grammar_manifest.all) |g| { if (grammarSelected(g)) count += 1; } return count; } const typst_supplement = \\ \\(call item: (ident) @keyword) \\(call item: (field field: (ident) @keyword)) \\"#" @keyword \\ ; fn querySrc(comptime g: grammar_manifest.Grammar) []const u8 { const base = @field(ts_queries, g.name ++ "_highlights"); const source = if (comptime std.mem.eql(u8, g.name, "typst")) base ++ typst_supplement else base; return comptime colorQuery(source); } fn colorQuery(comptime source: []const u8) []const u8 { @setEvalBranchQuota(2_000_000); var result: [source.len]u8 = undefined; var written: usize = 0; var at: usize = 0; while (at < source.len) { while (at < source.len) { if (std.ascii.isWhitespace(source[at])) { at += 1; } else if (source[at] == ';') { while (at < source.len and source[at] != '\n') at += 1; } else break; } const begin = at; var depth: usize = 0; var colored = false; var complete = false; while (at < source.len) { const byte = source[at]; if (complete and (byte == '(' or byte == '[' or byte == '"')) break; if (byte == ';') { while (at < source.len and source[at] != '\n') at += 1; continue; } if (byte == '"') { at += 1; while (at < source.len) : (at += 1) { if (source[at] == '\\') { at += 1; } else if (source[at] == '"') { at += 1; break; } } if (depth == 0) complete = true; continue; } if (byte == '(' or byte == '[') depth += 1; if (byte == ')' or byte == ']') { depth -= 1; if (depth == 0) complete = true; } if (byte == '@') { const name = at + 1; at = name; while (at < source.len and (std.ascii.isAlphanumeric(source[at]) or source[at] == '_' or source[at] == '.' or source[at] == '-')) at += 1; colored = colored or synFor(source[name..at]) != .none; continue; } at += 1; } if (colored) { @memcpy(result[written..][0 .. at - begin], source[begin..at]); written += at - begin; } } const filtered = result[0..written].*; return &filtered; } fn initSpecs() [specCount()]Spec { var out: [specCount()]Spec = undefined; var i = 0; inline for (grammar_manifest.all) |g| { if (grammarSelected(g)) { out[i] = .{ .name = g.name, .exts = g.exts, .language = @extern(LanguageFn, .{ .name = "tree_sitter_" ++ g.name }), .query_src = querySrc(g), .objects_src = @field(ts_queries, g.name ++ "_textobjects"), }; i += 1; } } return out; } var specs = initSpecs(); /// What tree-sitter allocates from, the runtime and every grammar's external /// scanner (build.zig compiles them to use the runtime's allocator): /// `start`'s allocator, through the shared C heap. var syntax_allocator: std.mem.Allocator = undefined; const syntax_heap = c_heap.Heap(&syntax_allocator); var syntax_started = false; extern fn ts_set_allocator( new_malloc: ?*const fn (size: usize) callconv(.c) ?*anyopaque, new_calloc: ?*const fn (nmemb: usize, size: usize) callconv(.c) ?*anyopaque, new_realloc: ?*const fn (ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque, new_free: ?*const fn (ptr: ?*anyopaque) callconv(.c) void, ) void; pub fn start(gpa: std.mem.Allocator) void { if (comptime enabled) { std.debug.assert(!syntax_started); stop(); syntax_allocator = gpa; syntax_started = true; ts_set_allocator(syntax_heap.malloc, syntax_heap.calloc, syntax_heap.realloc, syntax_heap.free); } } pub fn stop() void { if (comptime enabled) { for (&specs) |*spec| { if (spec.selected) |selected| { selected.cursor.destroy(); selected.parser.destroy(); selected.query.destroy(); } spec.selected = null; if (spec.objects) |q| q.destroy(); spec.objects = null; spec.objects_failed = false; } dropDiffPieces(); ts_set_allocator(null, null, null, null); syntax_started = false; syntax_allocator = undefined; } } const Selected = struct { name: []const u8, lang: *const ts.Language, query: *ts.Query, parser: *ts.Parser, cursor: *ts.QueryCursor, capture_styles: []u8, }; fn ensure(spec: *Spec) !Selected { if (spec.selected) |selected| return selected; const lang = spec.language(); var error_offset: u32 = 0; const query = try ts.Query.create(lang, spec.query_src, &error_offset); errdefer query.destroy(); if (query.captureCount() > spec.capture_styles.len) return error.TooManyCaptures; const capture_styles = spec.capture_styles[0..query.captureCount()]; for (capture_styles, 0..) |*style, id| { const name = query.captureNameForId(@intCast(id)) orelse ""; style.* = @intFromEnum(synFor(name)); if (style.* == 0) query.disableCapture(name); } const parser = ts.Parser.create(); errdefer parser.destroy(); try parser.setLanguage(lang); const selected: Selected = .{ .name = spec.name, .lang = lang, .query = query, .parser = parser, .cursor = ts.QueryCursor.create(), .capture_styles = capture_styles, }; spec.selected = selected; return selected; } fn forExt(ext: []const u8) !?Selected { for (&specs) |*spec| { for (spec.exts) |choice| { if (std.ascii.eqlIgnoreCase(ext, choice)) return try ensure(spec); } } return null; } fn forLang(name: []const u8) !?Selected { const canonical = if (grammar_manifest.byName(name)) |i| grammar_manifest.all[i].name else name; for (&specs) |*spec| { if (std.ascii.eqlIgnoreCase(canonical, spec.name)) return try ensure(spec); } return null; } fn runQuery(styles: []u8, sel: Selected, tree: *ts.Tree, base: usize) void { const cursor = sel.cursor; cursor.exec(sel.query, tree.rootNode()); while (cursor.nextMatch()) |match| { for (match.captures) |cap| { const style = sel.capture_styles[cap.index]; if (style == 0) continue; const b = @min(base + cap.node.startByte(), styles.len); const end = @min(base + @as(usize, cap.node.endByte()), styles.len); if (end > b) @memset(styles[b..end], style); } } } fn synFor(name: []const u8) Syn { for ([_]struct { []const u8, Syn }{ .{ "comment", .comment }, .{ "string", .string }, .{ "character", .string }, .{ "number", .number }, .{ "numeric", .number }, .{ "float", .number }, .{ "boolean", .number }, .{ "keyword", .keyword }, .{ "include", .keyword }, .{ "conditional", .keyword }, .{ "repeat", .keyword }, .{ "title", .keyword }, .{ "uri", .string }, .{ "reference", .number }, .{ "heading", .keyword }, .{ "strong", .comment }, .{ "bold", .comment }, .{ "emphasis", .number }, .{ "italic", .number }, .{ "literal", .string }, .{ "raw", .string }, }) |m| { if (std.mem.indexOf(u8, name, m[0]) != null) return m[1]; } return .none; } pub fn highlightFileRange(gpa: std.mem.Allocator, path: []const u8, content: []const u8, start_raw: usize, end_raw: usize) ![]u8 { const zone = tracy.zone(@src(), "highlightFileRange"); defer zone.end(); if (!enabled) return &.{}; const selected = (try forExt(std.fs.path.extension(path))) orelse return &.{}; const start_byte = @min(start_raw, content.len); const end_byte = @max(start_byte, @min(end_raw, content.len)); const source = content[start_byte..end_byte]; const styles = try gpa.alloc(u8, source.len); @memset(styles, 0); paint(styles, source, selected); return styles; } /// Source spans are independent of editor rows. Locations output uses the same /// declarations as sticky editor headers. Byte ends are exclusive; rows are /// zero-based and inclusive, including the last signature/opening-brace row. pub const ContextDeclaration = struct { start_byte: usize, end_byte: usize, start_line: usize, end_line: usize, header_end_byte: usize, header_end_line: usize, }; pub fn supportsPath(path: []const u8) bool { if (!enabled) return false; const ext = std.fs.path.extension(path); for (specs) |spec| for (spec.exts) |choice| { if (std.ascii.eqlIgnoreCase(ext, choice)) return true; }; return false; } /// A whole file's parse, kept by the file for the keys that walk its nodes /// (helix's Alt-o and kin). Opaque outside this file; null without a grammar. pub fn parseTree(path: []const u8, content: []const u8) ?*anyopaque { if (!enabled) return null; const selected = (forExt(std.fs.path.extension(path)) catch return null) orelse return null; const tree = selected.parser.parseString(content, null) orelse return null; return @ptrCast(tree); } pub fn freeTree(tree: *anyopaque) void { if (!enabled) return; const t: *ts.Tree = @ptrCast(@alignCast(tree)); t.destroy(); } /// Where a node walk goes from a range (helix object.rs and movement.rs). pub const NodeWalk = @import("modal.zig").Normal.NodeWalk; pub const Span = struct { from: usize, to: usize }; /// The spans `walk` takes the byte range [from, to) to, into `out`; `at` is /// the range's cursor. None when the tree has nothing to say there. pub fn walkNodes(tree_ptr: *anyopaque, walk: NodeWalk, from: usize, to: usize, at: usize, out: []Span) usize { if (!enabled or out.len == 0) return 0; const tree: *const ts.Tree = @ptrCast(@alignCast(tree_ptr)); const lo: u32 = @intCast(@min(from, std.math.maxInt(u32))); const hi: u32 = @intCast(@min(to, std.math.maxInt(u32))); if (walk == .parent_end or walk == .parent_start) { var node = tree.rootNode().namedDescendantForByteRange(lo, hi) orelse return 0; if (walk == .parent_end) { out[0] = .{ .from = node.endByte(), .to = node.endByte() }; return 1; } // already at the node's lo: its first ancestor that starts earlier if (node.startByte() == at) { const first = node.startByte(); while (node.startByte() >= first or !node.isNamed()) node = node.parent() orelse break; } out[0] = .{ .from = node.startByte(), .to = node.startByte() }; return 1; } var cursor = tree.walk(); defer cursor.destroy(); // helix TreeCursor.reset_to_byte_range: the smallest node holding the range while (true) { const node = cursor.node(); if (lo < node.startByte() or hi > node.endByte()) { _ = cursor.gotoParent(); break; } if (cursor.gotoFirstChildForByte(lo) == null) break; } switch (walk) { .expand => while (cursor.node().startByte() == lo and cursor.node().endByte() == hi) { if (!cursor.gotoParent()) break; }, .shrink => _ = cursor.gotoFirstChild(), .next_sibling => while (!cursor.gotoNextSibling()) { if (!cursor.gotoParent()) break; }, .prev_sibling => while (!cursor.gotoPreviousSibling()) { if (!cursor.gotoParent()) break; }, .all_siblings, .all_children => { if (walk == .all_siblings) { while (true) { if (!cursor.gotoParent()) return 0; if (cursor.node().childCount() > 1) break; } } // the named children, or nothing to say var n: usize = 0; if (!cursor.gotoFirstChild()) return 0; while (true) { const child = cursor.node(); if (child.isNamed() and n < out.len) { out[n] = .{ .from = child.startByte(), .to = child.endByte() }; n += 1; } if (!cursor.gotoNextSibling()) break; } return n; }, .parent_end, .parent_start => unreachable, } const node = cursor.node(); out[0] = .{ .from = node.startByte(), .to = node.endByte() }; return 1; } /// helix's textobjects, which its `]f`, `mif` and kin select: the name a /// query's captures begin with (`function.around`, `test.inside`). pub const Object = @import("modal.zig").Normal.Object; fn objectName(o: Object) []const u8 { return if (o == .xml_element) "xml-element" else @tagName(o); } /// The grammar's textobject query for `path`, compiled the first time; null /// when the grammar has none. fn objectQuery(path: []const u8) ?*ts.Query { const ext = std.fs.path.extension(path); for (&specs) |*spec| for (spec.exts) |choice| { if (!std.ascii.eqlIgnoreCase(ext, choice)) continue; if (spec.objects) |q| return q; if (spec.objects_src.len == 0 or spec.objects_failed) return null; var error_offset: u32 = 0; spec.objects = ts.Query.create(spec.language(), spec.objects_src, &error_offset) catch { spec.objects_failed = true; return null; }; return spec.objects; }; return null; } fn captureId(q: *const ts.Query, name: []const u8) ?u32 { for (0..q.captureCount()) |i| if (std.mem.eql(u8, q.captureNameForId(@intCast(i)) orelse "", name)) return @intCast(i); return null; } /// A match's `capture`, as one span over all its nodes (helix's grouped /// capture), when the match's `#eq?` and `#match?` hold. fn matchSpan(q: *const ts.Query, match: anytype, capture: u32, text: []const u8) ?Span { var span: ?Span = null; for (match.captures) |c| if (c.index == capture) { const s: Span = .{ .from = c.node.startByte(), .to = c.node.endByte() }; span = if (span) |old| .{ .from = @min(old.from, s.from), .to = @max(old.to, s.to) } else s; }; var found = span orelse return null; // a node that takes its line's newline (this zig grammar's comments) // is the line without it, as helix's grammars have it if (found.to > found.from + 1 and found.to <= text.len and text[found.to - 1] == '\n') found.to -= 1; const steps = q.predicatesForPattern(match.pattern_index); var i: usize = 0; while (i < steps.len) { var end = i; while (end < steps.len and steps[end].type != .done) end += 1; const pred = steps[i..end]; i = end + 1; if (pred.len != 3 or pred[0].type != .string or pred[1].type != .capture) continue; const op = q.stringValueForId(pred[0].value_id) orelse continue; const subject = for (match.captures) |c| { if (c.index == pred[1].value_id) break text[@min(c.node.startByte(), text.len)..@min(c.node.endByte(), text.len)]; } else continue; const want = if (pred[2].type == .string) q.stringValueForId(pred[2].value_id) orelse continue else for (match.captures) |c| { if (c.index == pred[2].value_id) break text[@min(c.node.startByte(), text.len)..@min(c.node.endByte(), text.len)]; } else continue; if (std.mem.eql(u8, op, "eq?")) { if (!std.mem.eql(u8, subject, want)) return null; } else if (std.mem.eql(u8, op, "match?")) { var re = regexp.Regex.compile(want) catch continue; if ((re.find(subject, 0, subject.len, subject.len) catch null) == null) return null; } } return found; } /// `mi`/`ma`: the smallest `object.inside` (or `.around`) holding byte `at` /// (helix textobject_treesitter). None without a query or such an object. pub fn objectAt(path: []const u8, tree_ptr: *anyopaque, text: []const u8, object: Object, around: bool, at: usize) ?Span { if (!enabled) return null; const q = objectQuery(path) orelse return null; var buf: [32]u8 = undefined; const want = std.fmt.bufPrint(&buf, "{s}.{s}", .{ objectName(object), if (around) "around" else "inside" }) catch return null; const capture = captureId(q, want) orelse return null; const tree: *const ts.Tree = @ptrCast(@alignCast(tree_ptr)); const cursor = ts.QueryCursor.create(); defer cursor.destroy(); cursor.exec(q, tree.rootNode()); var best: ?Span = null; while (cursor.nextMatch()) |match| { const s = matchSpan(q, match, capture, text) orelse continue; if (!(s.from <= at and at < s.to)) continue; if (best == null or s.to - s.from < best.?.to - best.?.from) best = s; } const s = best orelse return null; if (s.from >= text.len or s.to >= text.len) return null; return s; } /// `]f`/`[f` and kin: the next object starting after byte `at`, or the /// previous one ending before it (helix goto_treesitter_object), from the /// first of `.movement`, `.around`, `.inside` the query names. pub fn objectNext(path: []const u8, tree_ptr: *anyopaque, text: []const u8, object: Object, forward: bool, at: usize) ?Span { if (!enabled) return null; const q = objectQuery(path) orelse return null; var buf: [32]u8 = undefined; const capture = for ([_][]const u8{ "movement", "around", "inside" }) |kind| { const want = std.fmt.bufPrint(&buf, "{s}.{s}", .{ objectName(object), kind }) catch return null; if (captureId(q, want)) |id| break id; } else return null; const tree: *const ts.Tree = @ptrCast(@alignCast(tree_ptr)); const cursor = ts.QueryCursor.create(); defer cursor.destroy(); cursor.exec(q, tree.rootNode()); var best: ?Span = null; while (cursor.nextMatch()) |match| { const s = matchSpan(q, match, capture, text) orelse continue; if (forward) { if (s.from <= at) continue; if (best == null or s.from < best.?.from or (s.from == best.?.from and s.to > best.?.to)) best = s; } else { if (s.to >= at) continue; if (best == null or s.to > best.?.to or (s.to == best.?.to and s.from < best.?.from)) best = s; } } const s = best orelse return null; if (s.from >= text.len or s.to >= text.len) return null; return s; } test "every vendored textobject query compiles against its grammar" { if (!enabled) return; var failed: usize = 0; for (&specs) |*spec| { if (spec.objects_src.len == 0) continue; var error_offset: u32 = 0; const q = ts.Query.create(spec.language(), spec.objects_src, &error_offset) catch { std.debug.print("textobjects for {s} do not compile (offset {d})\n", .{ spec.name, error_offset }); failed += 1; continue; }; q.destroy(); } try std.testing.expectEqual(@as(usize, 0), failed); } /// Owned source analysis. Both slices use the allocator passed to analyzeSource. /// The parsed tree is released before returning; callers can cache this result. pub const SourceAnalysis = struct { declarations: []ContextDeclaration = &.{}, colors: []u8 = &.{}, pub fn deinit(self: *SourceAnalysis, gpa: std.mem.Allocator) void { gpa.free(self.declarations); gpa.free(self.colors); self.* = .{}; } }; pub fn analyzeSource(gpa: std.mem.Allocator, path: []const u8, content: []const u8, want_declarations: bool, want_colors: bool) !SourceAnalysis { if (!enabled or content.len == 0 or (!want_declarations and !want_colors)) return .{}; const selected = (try forExt(std.fs.path.extension(path))) orelse return .{}; const tree = selected.parser.parseString(content, null) orelse return .{}; defer tree.destroy(); var result: SourceAnalysis = .{}; errdefer result.deinit(gpa); if (want_declarations) result.declarations = try declarationsFromTree(gpa, content, tree); if (want_colors) { result.colors = try gpa.alloc(u8, content.len); @memset(result.colors, 0); paintTree(result.colors, content, selected, tree); } return result; } pub fn contextDeclarations(gpa: std.mem.Allocator, path: []const u8, content: []const u8) ![]ContextDeclaration { const analysis = try analyzeSource(gpa, path, content, true, false); return analysis.declarations; } fn declarationsFromTree(gpa: std.mem.Allocator, content: []const u8, tree: *ts.Tree) ![]ContextDeclaration { var result: std.ArrayList(ContextDeclaration) = .empty; errdefer result.deinit(gpa); // The cursor retains its ancestor stack: sibling visits do not repeatedly // reconstruct parents in broad syntax trees. Preorder preserves source order. var cursor = tree.rootNode().walk(); defer cursor.destroy(); while (true) { const node = cursor.node(); if (node.isNamed()) if (contextSpan(node, content)) |span| { if (result.items.len == 0 or result.items[result.items.len - 1].start_line != span.start_line or result.items[result.items.len - 1].end_line != span.end_line) try result.append(gpa, span); }; if (cursor.gotoFirstChild()) continue; while (!cursor.gotoNextSibling()) { if (!cursor.gotoParent()) return result.toOwnedSlice(gpa); } } } fn contextSpan(node: ts.Node, source: []const u8) ?ContextDeclaration { const kind = node.kind(); var body: ?ts.Node = null; var accepted = false; if (std.mem.eql(u8, kind, "Decl")) { // Zig's FnProto excludes its body: the surrounding Decl owns both. accepted = childOfKind(node, "FnProto") != null; body = childOfKind(node, "Block"); if (body == null) return null; } else if (std.mem.eql(u8, kind, "ContainerDecl") or std.mem.eql(u8, kind, "TestDecl")) { accepted = true; body = childOfKind(node, "Block"); } else { const kinds = [_][]const u8{ "function_definition", "function_declaration", "function_item", "function", "method_definition", "method_declaration", "method", "singleton_method", "constructor_declaration", "destructor_definition", "class_definition", "class_declaration", "class_specifier", "class", "singleton_class", "object_declaration", "struct_item", "struct_specifier", "struct_declaration", "union_specifier", "union_item", "union_declaration", "enum_item", "enum_specifier", "enum_declaration", "interface_declaration", "trait_item", "trait_definition", "impl_item", "mod_item", "module_definition", "module_declaration", "module", "namespace_definition", "namespace_declaration", "record_declaration", "type_declaration", "extension_declaration", "protocol_declaration", }; for (kinds) |candidate| if (std.mem.eql(u8, kind, candidate)) { accepted = true; break; }; body = node.childByFieldName("body"); } if (!accepted) return null; var owner = node; if (std.mem.eql(u8, kind, "ContainerDecl")) { // A multiline `const Name = struct` starts at the binding, not at the // struct token. Stop at real scopes so returned anonymous containers // do not inherit an enclosing function's signature. var ancestor = node.parent(); while (ancestor) |parent| { const parent_kind = parent.kind(); if (std.mem.eql(u8, parent_kind, "VarDecl")) { owner = parent; break; } if (std.mem.eql(u8, parent_kind, "Block") or std.mem.eql(u8, parent_kind, "Decl") or std.mem.eql(u8, parent_kind, "ContainerDecl") or std.mem.eql(u8, parent_kind, "FnProto") or std.mem.eql(u8, parent_kind, "ContainerField") or std.mem.eql(u8, parent_kind, "ParamDecl")) break; ancestor = parent.parent(); } } const beginning = owner.startPoint(); const end = node.endPoint(); if (beginning.row >= end.row) return null; var header_end: usize = node.startByte(); if (body) |block| { header_end = block.startByte(); // Indentation grammars start their body at its first statement. if (header_end < source.len and source[header_end] == '{') { header_end += 1; } else { while (header_end > node.startByte() and std.ascii.isWhitespace(source[header_end - 1])) header_end -= 1; } } else { // Containers often expose their braces directly, without a body node. var i: u32 = 0; while (i < node.childCount()) : (i += 1) { const child = node.child(i) orelse continue; if (std.mem.eql(u8, child.kind(), "{")) { header_end = child.endByte(); break; } } // Ruby and similar grammars have a body_statement without a field. if (header_end == node.startByte()) { header_end = std.mem.indexOfScalarPos(u8, source, header_end, '\n') orelse source.len; } } const line_start = @as(usize, owner.startByte()) - beginning.column; const header_line = beginning.row + std.mem.count(u8, source[line_start..@min(header_end, source.len)], "\n"); const header_line_end = std.mem.indexOfScalarPos(u8, source, @min(header_end, source.len), '\n') orelse source.len; return .{ .start_byte = line_start, .end_byte = node.endByte(), .start_line = beginning.row, .end_line = end.row - @as(usize, if (end.column == 0) 1 else 0), .header_end_byte = header_line_end, .header_end_line = header_line, }; } fn paint(styles: []u8, source: []const u8, selected: Selected) void { const tree = selected.parser.parseString(source, null) orelse return; defer tree.destroy(); paintTree(styles, source, selected, tree); } fn paintTree(styles: []u8, source: []const u8, selected: Selected, tree: *ts.Tree) void { runQuery(styles, selected, tree, 0); if (std.mem.eql(u8, selected.name, "markdown")) { inject(styles, source, tree.rootNode(), true); } else if (std.mem.eql(u8, selected.name, "typst")) { inject(styles, source, tree.rootNode(), false); } } test "syntax context nested Zig containers and multiline function signatures" { if (!enabled) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const source = "pub const Outer = struct {\n" ++ " const Inner = struct {\n" ++ " pub fn run(\n" ++ " value: u32,\n" ++ " ) void {\n" ++ " const ignored = .{\n" ++ " value,\n" ++ " };\n" ++ " }\n" ++ " };\n" ++ "};\n"; const declarations = try contextDeclarations(gpa, "nested.zig", source); defer gpa.free(declarations); try std.testing.expectEqual(@as(usize, 3), declarations.len); try std.testing.expectEqual(@as(usize, 0), declarations[0].start_line); try std.testing.expectEqual(@as(usize, 10), declarations[0].end_line); try std.testing.expectEqual(@as(usize, 1), declarations[1].start_line); try std.testing.expectEqual(@as(usize, 2), declarations[2].start_line); try std.testing.expectEqual(@as(usize, 4), declarations[2].header_end_line); try std.testing.expectEqual(@as(usize, 8), declarations[2].end_line); try std.testing.expectEqualStrings(" pub fn run(\n value: u32,\n ) void {", source[declarations[2].start_byte..declarations[2].header_end_byte]); const unsupported = try contextDeclarations(gpa, "notes.unknown", source); defer gpa.free(unsupported); try std.testing.expectEqual(@as(usize, 0), unsupported.len); } test "syntax context Python class and function body excludes first statement" { if (!enabled or !full_grammars) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const source = "class Outer:\n def run(\n self, value,\n ):\n return value\n"; const declarations = try contextDeclarations(gpa, "nested.py", source); defer gpa.free(declarations); try std.testing.expectEqual(@as(usize, 2), declarations.len); try std.testing.expectEqual(@as(usize, 0), declarations[0].header_end_line); try std.testing.expectEqual(@as(usize, 3), declarations[1].header_end_line); } test "syntax context multiline bindings and C++ namespaces preserve declaration starts" { if (!enabled) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const zig_source = "const Outer =\n struct {\n field: u8,\n };\n"; const zig_declarations = try contextDeclarations(gpa, "nested.zig", zig_source); defer gpa.free(zig_declarations); try std.testing.expectEqual(@as(usize, 1), zig_declarations.len); try std.testing.expectEqual(@as(usize, 0), zig_declarations[0].start_line); try std.testing.expectEqual(@as(usize, 1), zig_declarations[0].header_end_line); if (!minimal_grammars and !full_grammars) return; const cpp_source = "namespace example {\nstruct Outer {\n int run() {\n return 1;\n }\n};\n}\n"; const cpp_declarations = try contextDeclarations(gpa, "nested.cpp", cpp_source); defer gpa.free(cpp_declarations); try std.testing.expectEqual(@as(usize, 3), cpp_declarations.len); for (cpp_declarations, 0..) |declaration, row| { try std.testing.expectEqual(row, declaration.start_line); try std.testing.expectEqual(row, declaration.header_end_line); } } test "syntax context Rust modules impls and methods remain nested" { if (!enabled or !full_grammars) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const source = "mod outer {\n impl Example {\n fn run(&self) {\n work();\n }\n }\n}\n"; const declarations = try contextDeclarations(gpa, "nested.rs", source); defer gpa.free(declarations); try std.testing.expectEqual(@as(usize, 3), declarations.len); for (declarations, 0..) |declaration, row| { try std.testing.expectEqual(row, declaration.start_line); try std.testing.expectEqual(row, declaration.header_end_line); try std.testing.expectEqual(@as(usize, 6) - row, declaration.end_line); } } pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byte_raw: usize, end_byte_raw: usize) ![]u8 { const tz = tracy.zone(@src(), "highlightLocations"); defer tz.end(); const start_byte = @min(start_byte_raw, content.len); const end_byte = @max(start_byte, @min(end_byte_raw, content.len)); const source = content[start_byte..end_byte]; if (!enabled) return &.{}; const styles = try gpa.alloc(u8, source.len); errdefer gpa.free(styles); @memset(styles, 0); var memo_ext: []const u8 = "\x00"; var memo: ?Selected = null; var painted = false; var offset: usize = 0; var lines = std.mem.splitScalar(u8, source, '\n'); while (lines.next()) |line| { defer offset += line.len + 1; const code = codeAfterLocation(line) orelse continue; const ext = std.fs.path.extension(code.path); if (!std.mem.eql(u8, ext, memo_ext)) { memo_ext = ext; memo = forExt(ext) catch null; } const selected = memo orelse continue; paint(styles[offset + code.at ..][0..code.text.len], code.text, selected); painted = true; } if (!painted) { gpa.free(styles); return &.{}; } return styles; } /// Formatted results carry the exact source boundary independently of their /// padded location column. This also colors context whose location is hidden. /// Consecutive source rows share one parse, retaining multiline syntax. pub fn highlightLocationRows(gpa: std.mem.Allocator, content: []const u8, rows: anytype) ![]u8 { if (!enabled or content.len == 0 or rows.len == 0) return &.{}; const styles = try gpa.alloc(u8, content.len); errdefer gpa.free(styles); @memset(styles, 0); const code = try gpa.alloc(u8, content.len); defer gpa.free(code); const colors = try gpa.alloc(u8, content.len); defer gpa.free(colors); const SourceRow = struct { index: usize, start: usize, len: usize }; const sources = try gpa.alloc(SourceRow, rows.len); defer gpa.free(sources); var source_count: usize = 0; var offset: usize = 0; for (rows, 0..) |row, index| { if (offset >= content.len) break; const end = std.mem.indexOfScalarPos(u8, content, offset, '\n') orelse content.len; const source_start = offset + @min(row.code_start, end - offset); var address_only = false; if (comptime @hasField(@TypeOf(row), "location_end")) { address_only = row.location_end > 0 and source_start == end; } if (!address_only) { sources[source_count] = .{ .index = index, .start = source_start, .len = end - source_start }; source_count += 1; } offset = @min(end + 1, content.len); } var painted = false; var first: usize = 0; while (first < source_count) { const first_row = rows[sources[first].index]; var end = first + 1; while (end < source_count) : (end += 1) { const row = rows[sources[end].index]; const previous = rows[sources[end - 1].index]; if (!std.mem.eql(u8, row.path, first_row.path) or row.at.line != previous.at.line +| 1) break; } const language = (forExt(std.fs.path.extension(first_row.path)) catch null); if (language) |selected| { var code_len: usize = 0; for (sources[first..end]) |source| { @memcpy(code[code_len..][0..source.len], content[source.start..][0..source.len]); code_len += source.len; if (code_len < code.len) { code[code_len] = '\n'; code_len += 1; } } if (code_len > 0) { @memset(colors[0..code_len], 0); paint(colors[0..code_len], code[0..code_len], selected); painted = true; var at: usize = 0; for (sources[first..end]) |source| { @memcpy(styles[source.start..][0..source.len], colors[at..][0..source.len]); at += source.len + 1; } } } first = end; } // A shown context range can begin inside a comment/string whose opener was // omitted. Formatter snapshots from the complete source take precedence, // including zero styles that remove misleading fragment-parser captures. if (comptime @hasField(@TypeOf(rows[0]), "colors")) { offset = 0; for (rows) |row| { if (offset >= content.len) break; const end = std.mem.indexOfScalarPos(u8, content, offset, '\n') orelse content.len; const source_start = offset + @min(row.code_start, end - offset); const len = @min(row.colors.len, end - source_start); if (len > 0) { @memcpy(styles[source_start..][0..len], row.colors[0..len]); painted = true; } offset = @min(end + 1, content.len); } } // Declaration context is intentionally muted by the output painter. Keep // its source in parsing groups, then remove every syntax/style flag after // complete-source snapshots have been applied. if (comptime @hasField(@TypeOf(rows[0]), "declaration")) { offset = 0; for (rows) |row| { if (offset >= content.len) break; const end = std.mem.indexOfScalarPos(u8, content, offset, '\n') orelse content.len; if (row.declaration) { const source_start = offset + @min(row.code_start, end - offset); @memset(styles[source_start..end], 0); } offset = @min(end + 1, content.len); } } if (!painted) { gpa.free(styles); return &.{}; } return styles; } fn codeAfterLocation(line: []const u8) ?struct { path: []const u8, at: usize, text: []const u8 } { const token_end = std.mem.indexOfAny(u8, line, " \t") orelse return null; if (token_end == 0) return null; const token = line[0..token_end]; const target = look.parsePathLine(token); if (target.end != token.len) return null; if (target.at.line == 0) return null; const at = token_end + 1; if (at >= line.len) return null; return .{ .path = target.path, .at = at, .text = line[at..] }; } test "syntax formatted locations color hidden context and preserve multiline source" { if (!enabled or (!minimal_grammars and !full_grammars)) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const Row = struct { path: []const u8, code_start: usize, at: look.Spot }; const first = "a path.c:1 \t/* open"; const second = " \t| comment body"; const third = "a path.c:3 \t*/ int value = 42;"; const content = first ++ "\n" ++ second ++ "\n" ++ third ++ "\n"; const rows = [_]Row{ .{ .path = "a path.c", .code_start = "a path.c:1 \t".len, .at = .{ .line = 1 } }, .{ .path = "a path.c", .code_start = " \t| ".len, .at = .{ .line = 2 } }, .{ .path = "a path.c", .code_start = "a path.c:3 \t".len, .at = .{ .line = 3 } }, }; const styles = try highlightLocationRows(gpa, content, &rows); defer gpa.free(styles); const body = std.mem.indexOf(u8, content, "comment body").?; try std.testing.expectEqual(@intFromEnum(Syn.comment), styles[body]); try std.testing.expectEqual(@intFromEnum(Syn.none), styles[body - 1]); const number = std.mem.indexOf(u8, content, "42").?; try std.testing.expectEqual(@intFromEnum(Syn.number), styles[number]); try std.testing.expectEqual(@intFromEnum(Syn.none), styles[0]); } test "syntax formatted locations separate alignment from Markdown indentation" { if (!enabled or !full_grammars) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const Row = struct { path: []const u8, code_start: usize, at: look.Spot }; const content = "short.md:1 \t# Heading\n" ++ "longer path.md:5 \t # Code\n"; const rows = [_]Row{ .{ .path = "short.md", .code_start = "short.md:1 \t".len, .at = .{ .line = 1 } }, .{ .path = "longer path.md", .code_start = "longer path.md:5 \t".len, .at = .{ .line = 5 } }, }; const styles = try highlightLocationRows(gpa, content, &rows); defer gpa.free(styles); try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[std.mem.indexOf(u8, content, "Heading").?]); try std.testing.expectEqual(@intFromEnum(Syn.string), styles[std.mem.indexOf(u8, content, "Code").?]); } fn inject(styles: []u8, source: []const u8, node: ts.Node, markdown: bool) void { const kind = node.kind(); if (markdown and std.mem.eql(u8, kind, "inline")) { const begin: usize = node.startByte(); const end: usize = node.endByte(); if (begin >= end or end > source.len) return; const text = source[begin..end]; // Every colored inline capture requires one of these delimiters. if (std.mem.indexOfAny(u8, text, "*_`[<\\\r\n") == null) return; const selected = (forLang("markdown_inline") catch return) orelse return; const tree = selected.parser.parseString(text, null) orelse return; defer tree.destroy(); runQuery(styles, selected, tree, begin); return; } if (std.mem.eql(u8, kind, if (markdown) "fenced_code_block" else "raw_blck")) { const lang = if (markdown) blk: { const info = childOfKind(node, "info_string") orelse return; break :blk childOfKind(info, "language") orelse return; } else node.childByFieldName("lang") orelse return; const content = childOfKind(node, if (markdown) "code_fence_content" else "blob") orelse return; const selected = (forLang(source[lang.startByte()..lang.endByte()]) catch return) orelse return; const begin: usize = content.startByte(); const end: usize = content.endByte(); if (begin > end or end > source.len) return; const tree = selected.parser.parseString(source[begin..end], null) orelse return; defer tree.destroy(); @memset(styles[begin..end], 0); runQuery(styles, selected, tree, begin); return; } var i: u32 = 0; const count = node.childCount(); while (i < count) : (i += 1) { if (node.child(i)) |child| inject(styles, source, child, markdown); } } fn childOfKind(node: ts.Node, kind: []const u8) ?ts.Node { var i: u32 = 0; const count = node.childCount(); while (i < count) : (i += 1) { if (node.child(i)) |c| { if (std.mem.eql(u8, c.kind(), kind)) return c; } } return null; } /// A diff line's style byte carries whether the line was added or removed /// over its syntax style (the low bits), so the painter can tint its row /// and mark its prefix. pub const diff_added: u8 = 0x80; pub const diff_removed: u8 = 0x40; pub const syn_bits: u8 = 0x3f; /// A file section's hunks are parsed together, each side as one text, in /// pieces: a piece takes whole hunks until it holds `diff_piece_lines` /// lines, and a hunk longer than `diff_piece_cut` is cut there, counted /// from the section's first hunk. So a view parses about what a file's /// view does (its rows and a margin), however many small hunks it shows, /// and a 20k-line new file is parsed where it is shown, not whole. A string /// or comment across a cut colours as a fragment would. const diff_piece_lines = 40; const diff_piece_cut = 80; /// Painted hunk pieces, keyed by their bytes and language: scrolling back /// over a hunk, or a terminal repainting the same output every frame, /// parses nothing again. Freed by `stop`. const DiffPiece = struct { hash: u64, styles: []u8, used: u64 }; var diff_pieces: [96]?DiffPiece = @splat(null); var diff_clock: u64 = 0; fn dropDiffPieces() void { for (&diff_pieces) |*slot| { if (slot.*) |piece| syntax_allocator.free(piece.styles); slot.* = null; } diff_clock = 0; } /// The styles of `content[start..end)` read as a unified diff: headers /// muted, `@@` lines as keywords, and each hunk's code in the language of /// the file its section names. A hunk's old side (context and removed /// lines) and new side (context and added lines) are each parsed whole, so /// a string or comment that spans lines colours as it does in the file. /// Added and removed lines carry `diff_added`/`diff_removed`. A file in no /// language pardes knows keeps the plain colours: added lines as strings, /// removed ones as numbers. pub fn highlightDiff(gpa: std.mem.Allocator, content: []const u8, start_raw: usize, end_raw: usize) ![]u8 { return highlightDiffRows(gpa, content, start_raw, end_raw, false); } /// The same for a terminal's rows, where a row the terminal wrapped /// continues the diff line above it. pub fn highlightDiffRows(gpa: std.mem.Allocator, content: []const u8, start_raw: usize, end_raw: usize, continued_rows: bool) ![]u8 { const tz = tracy.zone(@src(), "highlightDiff"); defer tz.end(); const start_byte = @min(start_raw, content.len); const end_byte = @max(start_byte, @min(end_raw, content.len)); const styles = try gpa.alloc(u8, end_byte - start_byte); errdefer gpa.free(styles); @memset(styles, 0); const Window = struct { styles: []u8, start: usize, fn put(w: @This(), at: usize, bytes: []const u8) void { const lo = @max(at, w.start); const hi = @min(at + bytes.len, w.start + w.styles.len); if (lo < hi) @memcpy(w.styles[lo - w.start .. hi - w.start], bytes[lo - at .. hi - at]); } fn fill(w: @This(), at: usize, len: usize, style: u8) void { const lo = @max(at, w.start); const hi = @min(at + len, w.start + w.styles.len); if (lo < hi) @memset(w.styles[lo - w.start .. hi - w.start], style); } }; const window: Window = .{ .styles = styles, .start = start_byte }; var walk: diff.Walk = .{ .continued_rows = continued_rows }; // The run of hunk lines not painted yet, with each line's kind. var piece_start: ?usize = null; var piece_end: usize = 0; var piece_path: []const u8 = ""; var kinds: [diff_piece_cut]diff.Kind = undefined; var nkinds: usize = 0; var at = diff.anchorBefore(content, start_byte); while (true) { const nl = std.mem.indexOfScalarPos(u8, content, at, '\n') orelse content.len; const got = walk.step(content[at..nl]); const body = switch (got.kind) { .context, .added, .removed, .no_newline => true, // The next hunk of the section goes on in the same piece. .hunk => piece_start != null and nkinds < diff_piece_lines, else => false, }; if (piece_start) |ps| if (!body or nkinds == diff_piece_cut) { if (piece_end >= start_byte) try paintDiffPiece(gpa, window, content[ps..piece_end], ps, kinds[0..nkinds], piece_path); piece_start = null; nkinds = 0; }; if (at >= end_byte and piece_start == null) break; if (body and (piece_start != null or got.kind != .hunk)) { if (piece_start == null) { piece_start = at; piece_path = walk.path(); } kinds[nkinds] = got.kind; nkinds += 1; piece_end = nl; } else if (nl >= start_byte) { const style: Syn = switch (got.kind) { .hunk => .keyword, .meta, .old_path, .new_path => .comment, else => .none, }; if (style != .none) window.fill(at, nl - at, @intFromEnum(style)); } if (nl >= content.len) break; at = nl + 1; } if (piece_start) |ps| if (piece_end >= start_byte) try paintDiffPiece(gpa, window, content[ps..piece_end], ps, kinds[0..nkinds], piece_path); return styles; } fn paintDiffPiece(gpa: std.mem.Allocator, window: anytype, bytes: []const u8, at: usize, kinds: []const diff.Kind, path: []const u8) !void { const selected: ?Selected = if (enabled and syntax_started) (forExt(std.fs.path.extension(path)) catch null) else null; if (selected) |sel| { const hash = std.hash.Wyhash.hash(std.hash.Wyhash.hash(0, sel.name), bytes); diff_clock += 1; for (&diff_pieces) |*slot| if (slot.*) |*piece| if (piece.hash == hash and piece.styles.len == bytes.len) { piece.used = diff_clock; return window.put(at, piece.styles); }; const colored = try syntax_allocator.alloc(u8, bytes.len); errdefer syntax_allocator.free(colored); try paintHunk(gpa, colored, bytes, kinds, sel); window.put(at, colored); var oldest: usize = 0; for (&diff_pieces, 0..) |*slot, i| { const piece = slot.* orelse { oldest = i; break; }; if (diff_pieces[oldest]) |o| if (piece.used < o.used) { oldest = i; }; } if (diff_pieces[oldest]) |old| syntax_allocator.free(old.styles); diff_pieces[oldest] = .{ .hash = hash, .styles = colored, .used = diff_clock }; return; } // No language: each line in one colour, as diffs were always shown. var lines = std.mem.splitScalar(u8, bytes, '\n'); var offset: usize = 0; for (kinds) |kind| { const line = lines.next() orelse break; const style: u8 = switch (kind) { .added => @intFromEnum(Syn.string) | diff_added, .removed => @intFromEnum(Syn.number) | diff_removed, .no_newline => @intFromEnum(Syn.comment), .hunk => @intFromEnum(Syn.keyword), else => 0, }; if (style != 0) window.fill(at + offset, line.len, style); offset += line.len + 1; } } /// One piece's styles: its old side and its new side each parsed as one /// text, and every line's code coloured from its side, behind its /// one-character prefix. A `@@` line between its hunks is in neither. fn paintHunk(gpa: std.mem.Allocator, out: []u8, bytes: []const u8, kinds: []const diff.Kind, sel: Selected) !void { @memset(out, 0); const old_src = try gpa.alloc(u8, bytes.len); defer gpa.free(old_src); const new_src = try gpa.alloc(u8, bytes.len); defer gpa.free(new_src); var old_len: usize = 0; var new_len: usize = 0; var lines = std.mem.splitScalar(u8, bytes, '\n'); for (kinds) |kind| { const line = lines.next() orelse break; const code = line[codeStart(line, kind)..]; // A row a terminal wrapped goes on the line above, as it was printed. const joined = continuesRow(line, kind); if (kind == .context or kind == .removed) { if (joined and old_len > 0) old_len -= 1; @memcpy(old_src[old_len..][0..code.len], code); old_len += code.len; old_src[old_len] = '\n'; old_len += 1; } if (kind == .context or kind == .added) { if (joined and new_len > 0) new_len -= 1; @memcpy(new_src[new_len..][0..code.len], code); new_len += code.len; new_src[new_len] = '\n'; new_len += 1; } } // A side is parsed only when a line takes its colours from it: the old // side for removed lines, the new side for added lines, and context // from the new side, else from the old. A piece that only adds or only // removes is one parse. var removes = false; var adds = false; for (kinds) |kind| { removes = removes or kind == .removed; adds = adds or kind == .added; } const context_new = adds or !removes; const old_colors = try gpa.alloc(u8, old_len); defer gpa.free(old_colors); @memset(old_colors, 0); if (!context_new or removes) paint(old_colors, old_src[0..old_len], sel); const new_colors = try gpa.alloc(u8, new_len); defer gpa.free(new_colors); @memset(new_colors, 0); if (context_new) paint(new_colors, new_src[0..new_len], sel); lines = std.mem.splitScalar(u8, bytes, '\n'); var offset: usize = 0; var old_at: usize = 0; var new_at: usize = 0; for (kinds) |kind| { const line = lines.next() orelse break; defer offset += line.len + 1; const skip = codeStart(line, kind); const code_len = line.len - skip; const dst = out[offset + skip ..][0..code_len]; if (continuesRow(line, kind)) { if ((kind == .context or kind == .removed) and old_at > 0) old_at -= 1; if ((kind == .context or kind == .added) and new_at > 0) new_at -= 1; } switch (kind) { .removed => @memcpy(dst, old_colors[old_at..][0..code_len]), .added => @memcpy(dst, new_colors[new_at..][0..code_len]), .context => @memcpy(dst, if (context_new) new_colors[new_at..][0..code_len] else old_colors[old_at..][0..code_len]), .no_newline => @memset(out[offset..][0..line.len], @intFromEnum(Syn.comment)), .hunk => @memset(out[offset..][0..line.len], @intFromEnum(Syn.keyword)), else => {}, } if (kind == .context or kind == .removed) old_at += code_len + 1; if (kind == .context or kind == .added) new_at += code_len + 1; const flag: u8 = switch (kind) { .added => diff_added, .removed => diff_removed, else => 0, }; if (flag != 0) for (out[offset..][0..line.len]) |*b| { b.* |= flag; }; } } fn continuesRow(line: []const u8, kind: diff.Kind) bool { return line.len > 0 and codeStart(line, kind) == 0 and (kind == .context or kind == .added or kind == .removed); } /// Where a hunk line's code starts: past its prefix, which a blank context /// line (its space stripped) and a terminal's wrapped row have none of. fn codeStart(line: []const u8, kind: diff.Kind) usize { if (line.len == 0) return 0; return switch (kind) { .context => @intFromBool(line[0] == ' '), .added => @intFromBool(line[0] == '+'), .removed => @intFromBool(line[0] == '-'), else => 0, }; } test "syntax a results row is coloured by the file its location names" { if (!enabled) return; start(std.testing.allocator); defer stop(); const gpa = std.testing.allocator; const content = "src/a.zig:1:1 const S = struct {};\n" ++ "src/b.md:2:1 # heading\n" ++ "just some prose with a colon: here\n" ++ "src/c.zig:3:1\n"; const styles = try highlightLocations(gpa, content, 0, content.len); defer gpa.free(styles); try std.testing.expectEqual(content.len, styles.len); for (styles[0.."src/a.zig:1:1".len]) |b| try std.testing.expectEqual(@as(u8, 0), b); const zig_kw = std.mem.indexOf(u8, content, "struct").?; try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw]); try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[zig_kw + 5]); const prose = std.mem.indexOf(u8, content, "just some prose").?; for (styles[prose .. prose + 14]) |b| try std.testing.expectEqual(@as(u8, 0), b); const bare = std.mem.indexOf(u8, content, "src/c.zig").?; for (styles[bare..]) |b| try std.testing.expectEqual(@as(u8, 0), b); } test "syntax a buffer with no locations in it keeps no highlights at all" { if (!enabled) return; start(std.testing.allocator); defer stop(); const content = "nothing has been said yet\n0: save: AccessDenied (x2)\n"; const styles = try highlightLocations(std.testing.allocator, content, 0, content.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(@as(usize, 0), styles.len); } test "syntax codeAfterLocation takes whole-token locations and nothing else" { const got = codeAfterLocation("src/x.zig:7:2-9 fn main() void {") orelse return error.ShouldBeALocation; try std.testing.expectEqualStrings("src/x.zig", got.path); try std.testing.expectEqualStrings("fn main() void {", got.text); try std.testing.expect(codeAfterLocation("main.zig some words") == null); try std.testing.expect(codeAfterLocation("note: this is prose") == null); try std.testing.expect(codeAfterLocation("src/x.zig:7:2x rest") == null); try std.testing.expect(codeAfterLocation("src/x.zig:7:2") == null); try std.testing.expect(codeAfterLocation("") == null); try std.testing.expect(codeAfterLocation(" leading space") == null); } test "syntax default full grammar set highlights Typst source" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = "// note\n#let answer = 42\n#let text = \"hello\"\n"; const styles = try highlightFileRange(std.testing.allocator, "paper.typst", source, 0, source.len); defer std.testing.allocator.free(styles); const comment_at = std.mem.indexOf(u8, source, "// note").?; const keyword_at = std.mem.indexOf(u8, source, "let").?; const number_at = std.mem.indexOf(u8, source, "42").?; const string_at = std.mem.indexOf(u8, source, "\"hello\"").?; try std.testing.expectEqual(Syn.comment, @as(Syn, @enumFromInt(styles[comment_at]))); try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(styles[keyword_at]))); try std.testing.expectEqual(Syn.number, @as(Syn, @enumFromInt(styles[number_at]))); try std.testing.expectEqual(Syn.string, @as(Syn, @enumFromInt(styles[string_at]))); const short_ext = try highlightFileRange(std.testing.allocator, "paper.typ", source, 0, source.len); defer std.testing.allocator.free(short_ext); try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(short_ext[keyword_at]))); } test "syntax Typst markup constructs paint and raw blocks inject their language" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = \\// note \\= Heading Title \\Some *bold* text with `raw` inline. \\#import "mod.typ": helper \\#let answer = 42 \\#emit(7, "arg") \\```zig \\fn widget() u8 { return 99; } \\``` \\ ; const styles = try highlightFileRange(std.testing.allocator, "paper.typ", source, 0, source.len); defer std.testing.allocator.free(styles); const synAt = struct { fn f(s: []const u8, src: []const u8, needle: []const u8, offset: usize) Syn { return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]); } }.f; try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "= Heading", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Heading", 0)); try std.testing.expectEqual(Syn.comment, synAt(styles, source, "bold", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "raw` inline", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "emit", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "#emit", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 3)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn widget", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "99", 0)); try std.testing.expectEqual(Syn.none, synAt(styles, source, "widget", 0)); try std.testing.expectEqual(Syn.comment, synAt(styles, source, "// note", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "import", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "let", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "42", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "\"arg\"", 0)); } test "syntax markdown highlights markup, injects inline spans and fenced code blocks" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = "# Heading *slant* Title\n" ++ "\n" ++ "Prose with **stout** and *lean* plus `snippet` inline.\n" ++ "\n" ++ "```zig\n" ++ "fn gadget() u8 { return 77; }\n" ++ "```\n"; const styles = try highlightFileRange(std.testing.allocator, "doc.md", source, 0, source.len); defer std.testing.allocator.free(styles); const synAt = struct { fn f(s: []const u8, src: []const u8, needle: []const u8, offset: usize) Syn { return @enumFromInt(s[std.mem.indexOf(u8, src, needle).? + offset]); } }.f; try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "Title", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "slant", 0)); try std.testing.expectEqual(Syn.comment, synAt(styles, source, "stout", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "lean", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "snippet", 0)); try std.testing.expectEqual(Syn.string, synAt(styles, source, "```zig", 0)); try std.testing.expectEqual(Syn.keyword, synAt(styles, source, "fn gadget", 0)); try std.testing.expectEqual(Syn.number, synAt(styles, source, "77", 0)); try std.testing.expectEqual(Syn.none, synAt(styles, source, "gadget", 0)); } test "syntax inline fast path agrees with full Markdown query" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const selected = (try forLang("markdown_inline")).?; const check = struct { fn compare(sel: Selected, source: []const u8) !void { const tree = sel.parser.parseString(source, null) orelse return error.ParseFailed; defer tree.destroy(); const expected = try std.testing.allocator.alloc(u8, source.len); defer std.testing.allocator.free(expected); const actual = try std.testing.allocator.alloc(u8, source.len); defer std.testing.allocator.free(actual); for ([_]Syn{ .none, .keyword }) |background| { @memset(expected, @intFromEnum(background)); @memset(actual, @intFromEnum(background)); runQuery(expected, sel, tree, 0); inject(actual, source, tree.rootNode(), true); if (!std.mem.eql(u8, expected, actual)) std.debug.print("inline mismatch: {s}\n", .{source}); try std.testing.expectEqualSlices(u8, expected, actual); } } }.compare; for ([_][]const u8{ "", "plain prose", "ação Ελληνικά 日本語 🙂", "123 456", "tabs\tand spaces", "'quoted' (parentheses) \"double quotes\"", "https://example.org a@b.org", "& ", "~~struck~~ $formula$", "*emphasis* __strong__", "**bold** _emphasis_", "`code` and ``a`b``", "[text](target \"title\")", "![description](image)", "[shortcut] [reference][label]", "[[wiki|text]]", " ", "text", "\\*escaped\\*", "soft\nline", "hard \nline", "hard\\\nline", "tab\t\nline", "hard \r\nline", "hard \rline", "**broken", "[broken](", "`broken", }) |source| try check(selected, source); for (0..128) |byte| { const char: u8 = @intCast(byte); const source = [_]u8{ char, 'a', 'b', char, ' ', char, char, 'c', char, char }; try check(selected, &source); } } test "syntax plain Markdown keeps block styles without starting the inline parser" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const source = "# Heading\n\nPlain prose.\n\n indented code\n"; const styles = try highlightFileRange(std.testing.allocator, "a.md", source, 0, source.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[2]); try std.testing.expectEqual(@intFromEnum(Syn.none), styles[std.mem.indexOf(u8, source, "Plain").?]); try std.testing.expectEqual(@intFromEnum(Syn.string), styles[std.mem.indexOf(u8, source, "indented").?]); for (specs) |spec| { if (std.mem.eql(u8, spec.name, "markdown_inline")) try std.testing.expect(spec.selected == null); } } test "syntax highlightDiff colors unified diff lines by prefix" { const diff_text = "diff --git a/x b/x\n" ++ "--- a/x\n" ++ "+++ b/x\n" ++ "@@ -1,2 +1,2 @@\n" ++ " context\n" ++ "-old line\n" ++ "+new line\n"; const styles = try highlightDiff(std.testing.allocator, diff_text, 0, diff_text.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "diff --git", 0)); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "--- a/x", 0)); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "+++ b/x", 0)); try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, diff_text, "@@ -1,2", 0)); try std.testing.expectEqual(Syn.none, diffSynAt(styles, diff_text, " context", 0)); // A file in no known language: each line in one colour, and tinted. try std.testing.expectEqual(Syn.number, diffSynAt(styles, diff_text, "-old line", 0)); try std.testing.expectEqual(Syn.string, diffSynAt(styles, diff_text, "+new line", 0)); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, "-old line").?] & diff_removed != 0); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, "+new line").? + 4] & diff_added != 0); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, " context").?] & (diff_added | diff_removed) == 0); } fn diffSynAt(styles: []const u8, text: []const u8, needle: []const u8, offset: usize) Syn { return @enumFromInt(styles[std.mem.indexOf(u8, text, needle).? + offset] & syn_bits); } test "syntax a Zig hunk is coloured as Zig, each side parsed whole" { if (!enabled) return; start(std.testing.allocator); defer stop(); const diff_text = "diff --git a/src/shape.zig b/src/shape.zig\n" ++ "index 1111111..2222222 100644\n" ++ "--- a/src/shape.zig\n" ++ "+++ b/src/shape.zig\n" ++ "@@ -1,6 +1,6 @@\n" ++ " const Shape = struct {\n" ++ "- side: u8 = 7,\n" ++ "+ side: u16 = 42,\n" ++ " const note =\n" ++ "- \\\\old words defer\n" ++ "+ \\\\new words defer\n" ++ " ;\n" ++ " };\n"; const styles = try highlightDiff(std.testing.allocator, diff_text, 0, diff_text.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, diff_text, "struct", 0)); try std.testing.expectEqual(Syn.number, diffSynAt(styles, diff_text, "7,", 0)); try std.testing.expectEqual(Syn.number, diffSynAt(styles, diff_text, "42", 0)); // A multiline string, told apart only by parsing the side it is on: // `defer` inside it is no keyword. try std.testing.expectEqual(Syn.string, diffSynAt(styles, diff_text, "old words defer", 10)); try std.testing.expectEqual(Syn.string, diffSynAt(styles, diff_text, "new words defer", 10)); // The prefix is no code: it carries the tint alone. const added = std.mem.indexOf(u8, diff_text, "+ side").?; try std.testing.expectEqual(diff_added, styles[added]); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, "42").?] & diff_added != 0); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, "7,").?] & diff_removed != 0); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "index 111", 0)); // A window in the middle is coloured as the whole is. const from = std.mem.indexOf(u8, diff_text, " const note").?; const part = try highlightDiff(std.testing.allocator, diff_text, from, diff_text.len); defer std.testing.allocator.free(part); try std.testing.expectEqualSlices(u8, styles[from..], part); } test "syntax a Python hunk is coloured as Python" { if (!enabled or !full_grammars) return; start(std.testing.allocator); defer stop(); const diff_text = "--- a/tool.py\n" ++ "+++ b/tool.py\n" ++ "@@ -1,3 +1,4 @@\n" ++ " def run(n):\n" ++ "- return n\n" ++ "+ # doubled now\n" ++ "+ return n * 2\n" ++ " \n"; const styles = try highlightDiff(std.testing.allocator, diff_text, 0, diff_text.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, diff_text, "def", 0)); try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, diff_text, "return n\n", 0)); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "# doubled", 2)); try std.testing.expectEqual(Syn.number, diffSynAt(styles, diff_text, "2\n", 0)); } test "syntax a deleted file's lines are coloured in its language" { if (!enabled) return; start(std.testing.allocator); defer stop(); const diff_text = "diff --git a/gone.zig b/gone.zig\n" ++ "deleted file mode 100644\n" ++ "--- a/gone.zig\n" ++ "+++ /dev/null\n" ++ "@@ -1,2 +0,0 @@\n" ++ "-const Gone = enum { a };\n" ++ "-// the end\n"; const styles = try highlightDiff(std.testing.allocator, diff_text, 0, diff_text.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, diff_text, "enum", 0)); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "// the end", 3)); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, "enum").?] & diff_removed != 0); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, diff_text, "deleted file", 0)); } test "syntax a hunk in no known language keeps the line colours" { if (!enabled) return; start(std.testing.allocator); defer stop(); const diff_text = "--- a/notes.unknownext\n" ++ "+++ b/notes.unknownext\n" ++ "@@ -1 +1 @@\n" ++ "-const struct 1\n" ++ "+const struct 2\n"; const styles = try highlightDiff(std.testing.allocator, diff_text, 0, diff_text.len); defer std.testing.allocator.free(styles); try std.testing.expectEqual(Syn.number, diffSynAt(styles, diff_text, "struct 1", 0)); try std.testing.expectEqual(Syn.string, diffSynAt(styles, diff_text, "struct 2", 0)); try std.testing.expectEqual(Syn.string, diffSynAt(styles, diff_text, "+const struct 2", 0)); } test "syntax git diff output in a command pane is told and coloured by its rows" { if (!enabled) return; start(std.testing.allocator); defer stop(); // A command pane's rows: its command line, then git's output, with one // long added line the terminal wrapped onto a second row. const rows = [_][]const u8{ "git diff", "diff --git a/main.zig b/main.zig", "--- a/main.zig", "+++ b/main.zig", "@@ -1,2 +1,2 @@", " pub fn main() void {", "+ const answer: u32 = 42; // the ", "wrapped rest", "- return;", "", }; try std.testing.expect(diff.looksLikeDiff(&rows)); try std.testing.expect(!diff.looksLikeDiff(rows[5..])); const text = try std.mem.join(std.testing.allocator, "\n", &rows); defer std.testing.allocator.free(text); const styles = try highlightDiffRows(std.testing.allocator, text, 0, text.len, true); defer std.testing.allocator.free(styles); try std.testing.expectEqual(Syn.none, diffSynAt(styles, text, "git diff", 0)); try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, text, "fn main", 0)); try std.testing.expectEqual(Syn.number, diffSynAt(styles, text, "42", 0)); // The wrapped row stays an added line, and inside the comment. try std.testing.expect(styles[std.mem.indexOf(u8, text, "wrapped rest").?] & diff_added != 0); try std.testing.expectEqual(Syn.comment, diffSynAt(styles, text, "wrapped rest", 0)); try std.testing.expect(styles[std.mem.indexOf(u8, text, "return").?] & diff_removed != 0); } test "syntax a section's small hunks share one parse" { if (!enabled) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const diff_text = "--- a/x.zig\n+++ b/x.zig\n" ++ "@@ -1 +1 @@\n-const a = 1;\n+const a = 2;\n" ++ "@@ -10 +10 @@\n-const b = 1;\n+const b = 2;\n" ++ "@@ -20 +20 @@\n-const c = 1;\n+const c = 2;\n" ++ "--- a/y.zig\n+++ b/y.zig\n" ++ "@@ -1 +1 @@\n-const d = 1;\n+const d = 2;\n"; const styles = try highlightDiff(gpa, diff_text, 0, diff_text.len); defer gpa.free(styles); var cached: usize = 0; for (diff_pieces) |slot| cached += @intFromBool(slot != null); try std.testing.expectEqual(@as(usize, 2), cached); // x.zig's three hunks, and y.zig's try std.testing.expectEqual(Syn.keyword, diffSynAt(styles, diff_text, "@@ -10", 0)); try std.testing.expectEqual(Syn.number, diffSynAt(styles, diff_text, "2;\n@@ -20", 0)); try std.testing.expect(styles[std.mem.indexOf(u8, diff_text, "@@ -10").?] & (diff_added | diff_removed) == 0); } test "syntax a hunk painted once is kept, and a long one is parsed in pieces" { if (!enabled) return; const gpa = std.testing.allocator; start(gpa); defer stop(); var text: std.Io.Writer.Allocating = .init(gpa); defer text.deinit(); const n = diff_piece_cut * 20 + 10; try text.writer.print("--- /dev/null\n+++ b/big.zig\n@@ -0,0 +1,{d} @@\n", .{n}); for (0..n) |i| try text.writer.print("+const v{d} = {d};\n", .{ i, i }); const content = text.written(); const at = std.mem.indexOf(u8, content, "+const v1000 ").?; const window = try highlightDiff(gpa, content, at, at + 200); defer gpa.free(window); try std.testing.expectEqual(Syn.number, @as(Syn, @enumFromInt(window["+const v1000 = ".len] & syn_bits))); var cached: usize = 0; for (diff_pieces) |slot| if (slot) |piece| { cached += 1; // One piece of many, not the whole hunk. try std.testing.expect(piece.styles.len < content.len / 4); }; try std.testing.expectEqual(@as(usize, 1), cached); const again = try highlightDiff(gpa, content, at, at + 200); defer gpa.free(again); try std.testing.expectEqualSlices(u8, window, again); cached = 0; for (diff_pieces) |slot| cached += @intFromBool(slot != null); try std.testing.expectEqual(@as(usize, 1), cached); } test "syntax result fragments preserve source indentation and inline markup" { if (!enabled) return; start(std.testing.allocator); defer stop(); const fixtures = [_]struct { path: []const u8, source: []const u8 }{ .{ .path = "a.zig", .source = " const number = 42; // note" }, .{ .path = "a.md", .source = "# Heading *slant*" }, .{ .path = "a.md", .source = "**bold** and `code`" }, .{ .path = "a.md", .source = " # this is indented code" }, .{ .path = "a.md", .source = "\t# tab-indented code" }, .{ .path = "a.py", .source = " return \"hello\"" }, }; for (fixtures) |fixture| { const expected = try highlightFileRange(std.testing.allocator, fixture.path, fixture.source, 0, fixture.source.len); defer std.testing.allocator.free(expected); const row = try std.fmt.allocPrint(std.testing.allocator, "{s}:12:3-9 {s}", .{ fixture.path, fixture.source }); defer std.testing.allocator.free(row); const actual = try highlightLocations(std.testing.allocator, row, 0, row.len); defer std.testing.allocator.free(actual); if (expected.len == 0) { try std.testing.expectEqual(@as(usize, 0), actual.len); continue; } const code_at = row.len - fixture.source.len; try std.testing.expectEqualSlices(u8, expected, actual[code_at..]); for (actual[0..code_at]) |style| try std.testing.expectEqual(@as(u8, 0), style); } } test "syntax query filtering preserves upstream colors" { if (!enabled) return; var allocator: std.heap.DebugAllocator(.{ .stack_trace_frames = 0, .safety = true }) = .init; defer if (allocator.deinit() != .ok) @panic("leaked syntax query allocations"); start(allocator.allocator()); defer stop(); const source = "// comment\n# Heading *inline*\nconst value = 42;\nif (true) { return \"quoted\"; }\n/* multi\nline */\n"; inline for (grammar_manifest.all) |grammar| { if (comptime grammarSelected(grammar)) { const selected = (try forLang(grammar.name)).?; const raw_source = @field(ts_queries, grammar.name ++ "_highlights") ++ (if (comptime std.mem.eql(u8, grammar.name, "typst")) typst_supplement else ""); var error_offset: u32 = 0; const raw_query = try ts.Query.create(selected.lang, raw_source, &error_offset); defer raw_query.destroy(); var reference = selected; reference.query = raw_query; reference.capture_styles = try std.testing.allocator.alloc(u8, raw_query.captureCount()); defer std.testing.allocator.free(reference.capture_styles); for (reference.capture_styles, 0..) |*style, id| style.* = @intFromEnum(synFor(raw_query.captureNameForId(@intCast(id)) orelse "")); const tree = selected.parser.parseString(source, null) orelse return error.ParseFailed; defer tree.destroy(); var expected: [source.len]u8 = @splat(0); var actual: [source.len]u8 = @splat(0); runQuery(&expected, reference, tree, 0); runQuery(&actual, selected, tree, 0); if (!std.mem.eql(u8, &expected, &actual)) std.debug.print("query mismatch: {s}\n", .{grammar.name}); try std.testing.expectEqualSlices(u8, &expected, &actual); } } } test "syntax Zig keyword captures cover both bytes beyond line ten thousand" { if (!enabled) return; start(std.testing.allocator); defer stop(); const code = "pub fn main() void {\n if (true) return;\n}\n"; const source = try std.testing.allocator.alloc(u8, 10_001 + code.len); defer std.testing.allocator.free(source); @memset(source[0..10_001], '\n'); @memcpy(source[10_001..], code); const styles = try highlightFileRange(std.testing.allocator, "a.zig", source, 10_001, source.len); defer std.testing.allocator.free(styles); for ([_][]const u8{ "fn", "if" }) |keyword| { const at = std.mem.indexOf(u8, code, keyword).?; try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[at]); try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[at + 1]); } } test "syntax allocator switching clears default-runtime caches" { if (!enabled) return; const source = "fn main() void {}"; const initial = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len); std.testing.allocator.free(initial); start(std.testing.allocator); const custom = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len); std.testing.allocator.free(custom); stop(); const restored = try highlightFileRange(std.testing.allocator, "a.zig", source, 0, source.len); defer std.testing.allocator.free(restored); defer stop(); try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[0]); try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[1]); } test "syntax location context snapshots retain omitted multiline comment scope" { if (!enabled or (!minimal_grammars and !full_grammars)) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const source = "/* documentation\nconst int value = 42;\n*/\n"; const source_colors = try highlightFileRange(gpa, "a.c", source, 0, source.len); defer gpa.free(source_colors); const code = "const int value = 42;"; const source_start = std.mem.indexOf(u8, source, code).?; const Row = struct { path: []const u8, code_start: usize, at: look.Spot, colors: []const u8 }; const content = "| \t" ++ code ++ "\n"; const rows = [_]Row{.{ .path = "a.c", .code_start = "| \t".len, .at = .{ .line = 2 }, .colors = source_colors[source_start..][0..code.len], }}; const styles = try highlightLocationRows(gpa, content, &rows); defer gpa.free(styles); for (styles[rows[0].code_start..][0..code.len]) |style| try std.testing.expectEqual(@intFromEnum(Syn.comment), style); try std.testing.expectEqual(@intFromEnum(Syn.none), styles[0]); } test "syntax declaration locations clear snapshots without muting ordinary context" { if (!enabled) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const header = "pub fn run() void {"; const content = "| \t" ++ header ++ "\n| \t const value = 42;\nfile.zig:3\t}\n"; const snapshots = [_]u8{0xff} ** header.len; const Row = struct { path: []const u8 = "file.zig", code_start: usize, at: look.Spot, colors: []const u8 = &.{}, declaration: bool = false, }; const rows = [_]Row{ .{ .code_start = 3, .at = .{ .line = 1 }, .colors = &snapshots, .declaration = true }, .{ .code_start = 3, .at = .{ .line = 2 } }, .{ .code_start = "file.zig:3\t".len, .at = .{ .line = 3 } }, }; const styles = try highlightLocationRows(gpa, content, &rows); defer gpa.free(styles); for (styles[3..][0..header.len]) |style| try std.testing.expectEqual(@as(u8, 0), style); try std.testing.expectEqual(@intFromEnum(Syn.number), styles[std.mem.indexOf(u8, content, "42").?]); } test "syntax stacked addresses do not split multiline source groups" { if (!enabled or (!minimal_grammars and !full_grammars)) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const Row = struct { path: []const u8 = "a.c", code_start: usize, location_end: usize, at: look.Spot }; const content = "a.c:1:1\n/* open\na.c:2:1\nint value = 42;\na.c:3:1\n*/ int other = 7;\n"; const rows = [_]Row{ .{ .code_start = 7, .location_end = 7, .at = .{ .line = 1 } }, .{ .code_start = 0, .location_end = 0, .at = .{ .line = 1 } }, .{ .code_start = 7, .location_end = 7, .at = .{ .line = 2 } }, .{ .code_start = 0, .location_end = 0, .at = .{ .line = 2 } }, .{ .code_start = 7, .location_end = 7, .at = .{ .line = 3 } }, .{ .code_start = 0, .location_end = 0, .at = .{ .line = 3 } }, }; const styles = try highlightLocationRows(gpa, content, &rows); defer gpa.free(styles); try std.testing.expectEqual(@intFromEnum(Syn.comment), styles[std.mem.indexOf(u8, content, "42").?]); try std.testing.expectEqual(@intFromEnum(Syn.number), styles[std.mem.indexOf(u8, content, "7;").?]); for ([_][]const u8{ "a.c:1:1", "a.c:2:1", "a.c:3:1" }) |label| { const at = std.mem.indexOf(u8, content, label).?; for (styles[at..][0..label.len]) |style| try std.testing.expectEqual(@intFromEnum(Syn.none), style); } } test "syntax source analysis preserves declarations and injected colors" { if (!enabled) return; const gpa = std.testing.allocator; start(gpa); defer stop(); const Fixture = struct { path: []const u8, source: []const u8 }; for ([_]Fixture{ .{ .path = "a.zig", .source = "const Outer = struct {\n // comment\n const Inner = struct {\n pub fn run(\n x: u32,\n ) u32 { return x; }\n };\n};\n" }, .{ .path = "a.cpp", .source = "namespace Outer {\nstruct Inner {\nint run(\n int x\n) { return x; }\n};\n}\n" }, .{ .path = "a.rs", .source = "mod outer {\nstruct Inner {}\nimpl Inner {\nfn run(\n &self\n) {}\n}\n}\n" }, .{ .path = "a.js", .source = "function outer() {\nclass Inner {\nrun(\n x\n) { return x; }\n}\n}\n" }, .{ .path = "a.py", .source = "class Outer:\n class Inner:\n def run(\n self, x\n ):\n return x\n" }, .{ .path = "a.md", .source = "# Heading\n\n```zig\npub fn run() void {}\n```\n" }, .{ .path = "a.typst", .source = "= Heading\n\n```zig\npub fn run() void {}\n```\n" }, }) |fixture| { if (!supportsPath(fixture.path)) continue; var analysis = try analyzeSource(gpa, fixture.path, fixture.source, true, true); defer analysis.deinit(gpa); const colors = try highlightFileRange(gpa, fixture.path, fixture.source, 0, fixture.source.len); defer gpa.free(colors); try std.testing.expectEqualSlices(u8, colors, analysis.colors); // Compare the previous named-node walk to the cursor walk, including // source ordering, wrapper deduplication and inclusive declaration ends. const selected = (try forExt(std.fs.path.extension(fixture.path))).?; const tree = selected.parser.parseString(fixture.source, null).?; defer tree.destroy(); var reference: std.ArrayList(ContextDeclaration) = .empty; defer reference.deinit(gpa); var node = tree.rootNode(); walk: while (true) { if (contextSpan(node, fixture.source)) |span| { if (reference.items.len == 0 or reference.items[reference.items.len - 1].start_line != span.start_line or reference.items[reference.items.len - 1].end_line != span.end_line) try reference.append(gpa, span); } if (node.namedChild(0)) |child| { node = child; continue; } while (node.nextNamedSibling() == null) node = node.parent() orelse break :walk; node = node.nextNamedSibling().?; } try std.testing.expectEqual(reference.items.len, analysis.declarations.len); for (reference.items, analysis.declarations) |old, new| try std.testing.expect(std.meta.eql(old, new)); var declarations_only = try analyzeSource(gpa, fixture.path, fixture.source, true, false); defer declarations_only.deinit(gpa); try std.testing.expectEqual(@as(usize, 0), declarations_only.colors.len); try std.testing.expectEqual(analysis.declarations.len, declarations_only.declarations.len); var colors_only = try analyzeSource(gpa, fixture.path, fixture.source, false, true); defer colors_only.deinit(gpa); try std.testing.expectEqual(@as(usize, 0), colors_only.declarations.len); try std.testing.expectEqualSlices(u8, analysis.colors, colors_only.colors); } var unsupported = try analyzeSource(gpa, "a.unknown", "text", true, true); unsupported.deinit(gpa); try std.testing.expectEqual(@as(usize, 0), unsupported.declarations.len); try std.testing.expectEqual(@as(usize, 0), unsupported.colors.len); }