From 2bdde8c5ec1049ea0f575c6f9165ed4cef00ea50 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 15 Sep 2026 13:21:29 -0300 Subject: Align location results and configure source context --- src/syntax.zig | 147 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 147 insertions(+) (limited to 'src/syntax.zig') diff --git a/src/syntax.zig b/src/syntax.zig index 69a69bda..8209dc2c 100644 --- a/src/syntax.zig +++ b/src/syntax.zig @@ -617,6 +617,86 @@ pub fn highlightLocations(gpa: std.mem.Allocator, content: []const u8, start_byt } return styles; } + +/// Formatted results carry the exact source boundary independently of their +/// padded location column. This also colors context whose location is hidden. +/// Consecutive source rows share one parse, retaining multiline syntax. +pub fn highlightLocationRows(gpa: std.mem.Allocator, content: []const u8, rows: anytype) ![]u8 { + if (!enabled or content.len == 0 or rows.len == 0) return &.{}; + const styles = try gpa.alloc(u8, content.len); + errdefer gpa.free(styles); + @memset(styles, 0); + const code = try gpa.alloc(u8, content.len); + defer gpa.free(code); + const colors = try gpa.alloc(u8, content.len); + defer gpa.free(colors); + var painted = false; + var row_index: usize = 0; + var offset: usize = 0; + while (row_index < rows.len and offset < content.len) { + const first = row_index; + const group_start = offset; + const selected = (forExt(std.fs.path.extension(rows[first].path)) catch null); + var code_len: usize = 0; + while (row_index < rows.len and offset < content.len) { + const row = rows[row_index]; + if (row_index > first and (!std.mem.eql(u8, row.path, rows[first].path) or + row.at.line != rows[row_index - 1].at.line +| 1)) break; + const end = std.mem.indexOfScalarPos(u8, content, offset, '\n') orelse content.len; + const source_start = offset + @min(row.code_start, end - offset); + if (row_index > first) { + code[code_len] = '\n'; + code_len += 1; + } + @memcpy(code[code_len..][0 .. end - source_start], content[source_start..end]); + code_len += end - source_start; + offset = @min(end + 1, content.len); + row_index += 1; + } + const language = selected orelse continue; + if (offset > 0 and content[offset - 1] == '\n') { + code[code_len] = '\n'; + code_len += 1; + } + if (code_len == 0) continue; + @memset(colors[0..code_len], 0); + paint(colors[0..code_len], code[0..code_len], language); + painted = true; + var target = group_start; + var source_offset: usize = 0; + for (rows[first..row_index]) |row| { + const end = std.mem.indexOfScalarPos(u8, content, target, '\n') orelse content.len; + const source_start = target + @min(row.code_start, end - target); + const len = end - source_start; + @memcpy(styles[source_start..end], colors[source_offset..][0..len]); + source_offset += len + 1; + target = @min(end + 1, content.len); + } + } + // A shown context range can begin inside a comment/string whose opener was + // omitted. Formatter snapshots from the complete source take precedence, + // including zero styles that remove misleading fragment-parser captures. + if (comptime @hasField(@TypeOf(rows[0]), "colors")) { + offset = 0; + for (rows) |row| { + if (offset >= content.len) break; + const end = std.mem.indexOfScalarPos(u8, content, offset, '\n') orelse content.len; + const source_start = offset + @min(row.code_start, end - offset); + const len = @min(row.colors.len, end - source_start); + if (len > 0) { + @memcpy(styles[source_start..][0..len], row.colors[0..len]); + painted = true; + } + offset = @min(end + 1, content.len); + } + } + if (!painted) { + gpa.free(styles); + return &.{}; + } + return styles; +} + fn codeAfterLocation(line: []const u8) ?struct { path: []const u8, at: usize, text: []const u8 } { const token_end = std.mem.indexOfAny(u8, line, " \t") orelse return null; if (token_end == 0) return null; @@ -628,6 +708,48 @@ fn codeAfterLocation(line: []const u8) ?struct { path: []const u8, at: usize, te if (at >= line.len) return null; return .{ .path = target.path, .at = at, .text = line[at..] }; } + +test "syntax formatted locations color hidden context and preserve multiline source" { + if (!enabled or (!minimal_grammars and !full_grammars)) return; + const gpa = std.testing.allocator; + start(gpa); + defer stop(); + const Row = struct { path: []const u8, code_start: usize, at: look.Spot }; + const first = "a path.c:1 \t/* open"; + const second = " \t| comment body"; + const third = "a path.c:3 \t*/ int value = 42;"; + const content = first ++ "\n" ++ second ++ "\n" ++ third ++ "\n"; + const rows = [_]Row{ + .{ .path = "a path.c", .code_start = "a path.c:1 \t".len, .at = .{ .line = 1 } }, + .{ .path = "a path.c", .code_start = " \t| ".len, .at = .{ .line = 2 } }, + .{ .path = "a path.c", .code_start = "a path.c:3 \t".len, .at = .{ .line = 3 } }, + }; + const styles = try highlightLocationRows(gpa, content, &rows); + defer gpa.free(styles); + const body = std.mem.indexOf(u8, content, "comment body").?; + try std.testing.expectEqual(@intFromEnum(Syn.comment), styles[body]); + try std.testing.expectEqual(@intFromEnum(Syn.none), styles[body - 1]); + const number = std.mem.indexOf(u8, content, "42").?; + try std.testing.expectEqual(@intFromEnum(Syn.number), styles[number]); + try std.testing.expectEqual(@intFromEnum(Syn.none), styles[0]); +} + +test "syntax formatted locations separate alignment from Markdown indentation" { + if (!enabled or !full_grammars) return; + const gpa = std.testing.allocator; + start(gpa); + defer stop(); + const Row = struct { path: []const u8, code_start: usize, at: look.Spot }; + const content = "short.md:1 \t# Heading\n" ++ "longer path.md:5 \t # Code\n"; + const rows = [_]Row{ + .{ .path = "short.md", .code_start = "short.md:1 \t".len, .at = .{ .line = 1 } }, + .{ .path = "longer path.md", .code_start = "longer path.md:5 \t".len, .at = .{ .line = 5 } }, + }; + const styles = try highlightLocationRows(gpa, content, &rows); + defer gpa.free(styles); + try std.testing.expectEqual(@intFromEnum(Syn.keyword), styles[std.mem.indexOf(u8, content, "Heading").?]); + try std.testing.expectEqual(@intFromEnum(Syn.string), styles[std.mem.indexOf(u8, content, "Code").?]); +} fn inject(styles: []u8, source: []const u8, node: ts.Node, markdown: bool) void { const kind = node.kind(); if (markdown and std.mem.eql(u8, kind, "inline")) { @@ -1056,3 +1178,28 @@ test "syntax allocator switching clears default-runtime caches" { try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[0]); try std.testing.expectEqual(@intFromEnum(Syn.keyword), restored[1]); } + +test "syntax location context snapshots retain omitted multiline comment scope" { + if (!enabled or (!minimal_grammars and !full_grammars)) return; + const gpa = std.testing.allocator; + start(gpa); + defer stop(); + const source = "/* documentation\nconst int value = 42;\n*/\n"; + const source_colors = try highlightFileRange(gpa, "a.c", source, 0, source.len); + defer gpa.free(source_colors); + const code = "const int value = 42;"; + const source_start = std.mem.indexOf(u8, source, code).?; + const Row = struct { path: []const u8, code_start: usize, at: look.Spot, colors: []const u8 }; + const content = "| \t" ++ code ++ "\n"; + const rows = [_]Row{.{ + .path = "a.c", + .code_start = "| \t".len, + .at = .{ .line = 2 }, + .colors = source_colors[source_start..][0..code.len], + }}; + const styles = try highlightLocationRows(gpa, content, &rows); + defer gpa.free(styles); + for (styles[rows[0].code_start..][0..code.len]) |style| + try std.testing.expectEqual(@intFromEnum(Syn.comment), style); + try std.testing.expectEqual(@intFromEnum(Syn.none), styles[0]); +} -- cgit v1.3