summaryrefslogtreecommitdiff
path: root/src/diff.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/diff.zig')
-rw-r--r--src/diff.zig337
1 files changed, 337 insertions, 0 deletions
diff --git a/src/diff.zig b/src/diff.zig
new file mode 100644
index 00000000..98942469
--- /dev/null
+++ b/src/diff.zig
@@ -0,0 +1,337 @@
+//! A unified diff (`git diff`, `diff -u`), read a line at a time: which
+//! file each section is about, what each line is, and where a hunk line
+//! stands in the old and the new file. The syntax painter (syntax.zig
+//! highlightDiff) walks it.
+//!
+//! A hunk's extent comes from the counts in its `@@` header, as patch
+//! reads it, so a removed line that itself starts `--` (`--- x`) is not
+//! taken for a file header.
+const std = @import("std");
+
+pub const Kind = enum {
+ /// Anything outside a file header and a hunk: a commit message,
+ /// `Only in ...`, a mail's headers.
+ other,
+ /// `diff ...`, `index ...`, `new file mode ...`, `rename from ...`,
+ /// `Binary files ... differ`: said of the section, naming no line.
+ meta,
+ /// `--- <old path>`
+ old_path,
+ /// `+++ <new path>`
+ new_path,
+ /// `@@ -a,b +c,d @@`
+ hunk,
+ context,
+ added,
+ removed,
+ /// `\ No newline at end of file`
+ no_newline,
+};
+
+/// One line, walked.
+pub const Line = struct {
+ kind: Kind,
+ /// Where it is in the old and the new file, from 1. A hunk header has
+ /// its first lines; a removed line the new line now standing where it
+ /// was (the one after it); an added line the old line after it.
+ old: usize = 0,
+ new: usize = 0,
+};
+
+pub const Walk = struct {
+ /// The section's paths, as written, borrowed from the lines walked:
+ /// `a/` and `b/` kept, timestamps cut.
+ old_path: []const u8 = "",
+ new_path: []const u8 = "",
+ /// `diff --git a/x b/y`'s b side, for a section with no `+++` yet
+ /// (a mode change, a rename, a binary file).
+ git_path: []const u8 = "",
+ old_left: usize = 0,
+ new_left: usize = 0,
+ old_at: usize = 0,
+ new_at: usize = 0,
+ /// The hunk's header numbers, which a look past its end clamps to.
+ new_start: usize = 0,
+ new_count: usize = 0,
+ /// Rows a terminal wrapped: inside a hunk, a row that starts with no
+ /// prefix at all continues the line above instead of ending the hunk.
+ continued_rows: bool = false,
+ last: Kind = .other,
+
+ pub fn inHunk(w: *const Walk) bool {
+ return w.old_left > 0 or w.new_left > 0;
+ }
+
+ /// The file the section is about: the new one, else (deleted) the old
+ /// one, else what its `diff --git` line names.
+ pub fn path(w: *const Walk) []const u8 {
+ if (w.new_path.len > 0 and !isDevNull(w.new_path)) return w.new_path;
+ if (w.old_path.len > 0 and !isDevNull(w.old_path)) return w.old_path;
+ return w.git_path;
+ }
+
+ /// Whether the section deletes its file (`+++ /dev/null`).
+ pub fn deleted(w: *const Walk) bool {
+ return isDevNull(w.new_path);
+ }
+
+ pub fn step(w: *Walk, line_raw: []const u8) Line {
+ const line = std.mem.trimEnd(u8, line_raw, "\r");
+ const got = w.stepInner(line);
+ w.last = got.kind;
+ return got;
+ }
+
+ fn stepInner(w: *Walk, line: []const u8) Line {
+ if (w.inHunk()) {
+ // A blank context line, its space stripped by a mailer or an
+ // editor, is still context.
+ const c: u8 = if (line.len == 0) ' ' else line[0];
+ switch (c) {
+ ' ' => if (w.old_left > 0 and w.new_left > 0) {
+ const at: Line = .{ .kind = .context, .old = w.old_at, .new = w.new_at };
+ w.old_left -= 1;
+ w.new_left -= 1;
+ w.old_at += 1;
+ w.new_at += 1;
+ return at;
+ },
+ '-' => if (w.old_left > 0) {
+ const at: Line = .{ .kind = .removed, .old = w.old_at, .new = w.new_at };
+ w.old_left -= 1;
+ w.old_at += 1;
+ return at;
+ },
+ '+' => if (w.new_left > 0) {
+ const at: Line = .{ .kind = .added, .old = w.old_at, .new = w.new_at };
+ w.new_left -= 1;
+ w.new_at += 1;
+ return at;
+ },
+ '\\' => return .{ .kind = .no_newline, .old = w.old_at, .new = w.new_at },
+ else => if (w.continued_rows and w.last != .hunk and w.last != .other)
+ return .{ .kind = w.last, .old = w.old_at, .new = w.new_at },
+ }
+ // Counts run out, or a line no hunk has: the hunk was cut short.
+ w.old_left = 0;
+ w.new_left = 0;
+ }
+ if (line.len > 0 and line[0] == '\\' and w.last != .other and w.last != .meta)
+ return .{ .kind = .no_newline, .old = w.old_at, .new = w.new_at };
+ if (std.mem.startsWith(u8, line, "diff ")) {
+ w.* = .{ .continued_rows = w.continued_rows };
+ w.git_path = gitPath(line);
+ return .{ .kind = .meta };
+ }
+ if (std.mem.startsWith(u8, line, "--- ")) {
+ // A second `---` with no `+++` between starts another section
+ // (plain `diff -u` output has no `diff` line of its own).
+ w.* = .{ .continued_rows = w.continued_rows, .git_path = w.git_path };
+ w.old_path = cutPath(line[4..]);
+ return .{ .kind = .old_path };
+ }
+ if (std.mem.startsWith(u8, line, "+++ ")) {
+ w.new_path = cutPath(line[4..]);
+ return .{ .kind = .new_path };
+ }
+ if (parseHunk(line)) |h| {
+ w.old_left = h.old_count;
+ w.new_left = h.new_count;
+ w.old_at = h.old_start;
+ w.new_at = h.new_start;
+ w.new_start = h.new_start;
+ w.new_count = h.new_count;
+ return .{ .kind = .hunk, .old = h.old_start, .new = h.new_start };
+ }
+ for (meta_words) |word| if (std.mem.startsWith(u8, line, word)) return .{ .kind = .meta };
+ return .{ .kind = .other };
+ }
+};
+
+const meta_words = [_][]const u8{
+ "index ", "new file mode ", "deleted file mode ", "old mode ", "new mode ",
+ "similarity index ", "rename from ", "rename to ", "copy from ", "copy to ",
+ "Binary files ", "dissimilarity ",
+};
+
+fn isDevNull(path: []const u8) bool {
+ return std.mem.eql(u8, path, "/dev/null");
+}
+
+/// A `---`/`+++` path with what follows it cut: `diff -u`'s tab and
+/// timestamp, and git's quotes round a name it had to escape.
+fn cutPath(rest: []const u8) []const u8 {
+ var name = rest[0 .. std.mem.indexOfScalar(u8, rest, '\t') orelse rest.len];
+ name = std.mem.trimEnd(u8, name, " \r");
+ if (name.len >= 2 and name[0] == '"' and name[name.len - 1] == '"') name = name[1 .. name.len - 1];
+ return name;
+}
+
+/// `diff --git a/x b/y`: the b side. A name with a space in it is split at
+/// the last ` b/`, which is right unless the old name has one too.
+fn gitPath(line: []const u8) []const u8 {
+ if (!std.mem.startsWith(u8, line, "diff --git ")) return "";
+ const rest = line["diff --git ".len..];
+ const at = std.mem.lastIndexOf(u8, rest, " b/") orelse return "";
+ return std.mem.trimEnd(u8, rest[at + 1 ..], " \r");
+}
+
+pub const HunkHeader = struct { old_start: usize, old_count: usize, new_start: usize, new_count: usize };
+
+/// `@@ -a[,b] +c[,d] @@ ...`, a count left out meaning one.
+pub fn parseHunk(line: []const u8) ?HunkHeader {
+ if (!std.mem.startsWith(u8, line, "@@ -")) return null;
+ var at: usize = "@@ -".len;
+ const old = range(line, &at) orelse return null;
+ if (at + 1 >= line.len or line[at] != ' ' or line[at + 1] != '+') return null;
+ at += 2;
+ const new = range(line, &at) orelse return null;
+ if (!std.mem.startsWith(u8, line[at..], " @@")) return null;
+ return .{ .old_start = old[0], .old_count = old[1], .new_start = new[0], .new_count = new[1] };
+}
+
+fn range(line: []const u8, at: *usize) ?[2]usize {
+ const start = number(line, at) orelse return null;
+ if (at.* < line.len and line[at.*] == ',') {
+ at.* += 1;
+ return .{ start, number(line, at) orelse return null };
+ }
+ return .{ start, 1 };
+}
+
+fn number(line: []const u8, at: *usize) ?usize {
+ const from = at.*;
+ var n: usize = 0;
+ while (at.* < line.len and std.ascii.isDigit(line[at.*])) : (at.* += 1)
+ n = std.math.add(usize, std.math.mul(usize, n, 10) catch return null, line[at.*] - '0') catch return null;
+ return if (at.* > from) n else null;
+}
+
+/// Where a walk that ends up at byte `pos` of `content` may safely start:
+/// the start of the nearest `diff ` line before it (no hunk line starts
+/// with a `d`), else the top.
+pub fn anchorBefore(content: []const u8, pos: usize) usize {
+ var end = @min(pos, content.len);
+ while (end > 0) {
+ const nl = std.mem.lastIndexOfScalar(u8, content[0..end], '\n') orelse break;
+ if (std.mem.startsWith(u8, content[nl + 1 ..], "diff ") and nl + 1 <= pos) return nl + 1;
+ end = nl;
+ }
+ return 0;
+}
+
+/// The same for a list of lines: the nearest `diff ` line at or above `row`.
+pub fn anchorRow(lines: []const []const u8, row: usize) usize {
+ var r = @min(row, lines.len);
+ while (r > 0) {
+ r -= 1;
+ if (std.mem.startsWith(u8, lines[r], "diff ")) return r;
+ }
+ return 0;
+}
+
+/// Whether text a command printed is a diff: a `diff --git` line, or a
+/// `--- ` line with a `+++ ` line under it, in its first rows (after
+/// `git show`'s commit header, say).
+pub fn looksLikeDiff(lines: []const []const u8) bool {
+ const n = @min(lines.len, 200);
+ for (lines[0..n], 0..) |line, i| {
+ if (std.mem.startsWith(u8, line, "diff --git ")) return true;
+ if (std.mem.startsWith(u8, line, "--- ") and i + 1 < lines.len and std.mem.startsWith(u8, lines[i + 1], "+++ ") and
+ i + 2 < lines.len and parseHunk(lines[i + 2]) != null) return true;
+ }
+ return false;
+}
+
+// ---- tests ----
+
+const testing = std.testing;
+
+fn splitLines(comptime text: []const u8) [std.mem.count(u8, text, "\n")][]const u8 {
+ @setEvalBranchQuota(100_000);
+ var out: [std.mem.count(u8, text, "\n")][]const u8 = undefined;
+ var it = std.mem.splitScalar(u8, text, '\n');
+ for (&out) |*line| line.* = it.next().?;
+ return out;
+}
+
+const git_diff =
+ \\diff --git a/src/a.zig b/src/a.zig
+ \\index 1111111..2222222 100644
+ \\--- a/src/a.zig
+ \\+++ b/src/a.zig
+ \\@@ -10,4 +10,4 @@ pub fn main() void {
+ \\ const a = 1;
+ \\-const b = 2;
+ \\+const b = 3;
+ \\+const c = 4;
+ \\ const d = 5;
+ \\--- x
+ \\@@ -40,2 +41,2 @@
+ \\ keep();
+ \\-gone();
+ \\+added();
+ \\diff --git a/old.py b/old.py
+ \\deleted file mode 100644
+ \\--- a/old.py
+ \\+++ /dev/null
+ \\@@ -1,2 +0,0 @@
+ \\-def f():
+ \\- return 1
+ \\
+;
+
+test "diff a walk follows the hunk counts, so a removed `--` line is no header" {
+ const lines = splitLines(git_diff);
+ var w: Walk = .{};
+ var kinds: [lines.len]Kind = undefined;
+ for (lines, 0..) |line, i| kinds[i] = w.step(line).kind;
+ try testing.expectEqualSlices(Kind, &.{
+ .meta, .meta, .old_path, .new_path, .hunk,
+ .context, .removed, .added, .added, .context,
+ .removed, .hunk, .context, .removed, .added,
+ .meta, .meta, .old_path, .new_path, .hunk,
+ .removed, .removed,
+ }, &kinds);
+}
+
+test "diff plain diff -u output: no a/ b/, timestamps after the names" {
+ const lines = splitLines("--- old/x.zig\t2026-09-30 10:00:00.000000000 +0000\n" ++
+ "+++ new/x.zig\t2026-09-30 11:00:00.000000000 +0000\n" ++
+ "@@ -3 +3,2 @@\n" ++
+ "-const a = 1;\n" ++
+ "+const a = 2;\n" ++
+ "+const b = 3;\n");
+ try testing.expect(looksLikeDiff(&lines));
+ var w: Walk = .{};
+ for (lines) |line| _ = w.step(line);
+ try testing.expectEqualStrings("new/x.zig", w.path());
+ try testing.expectEqualStrings("old/x.zig", w.old_path);
+}
+
+test "diff output is told from other text by its first rows" {
+ const git = [_][]const u8{ "$ git diff", "diff --git a/x b/x", "index 1..2" };
+ try testing.expect(looksLikeDiff(&git));
+ const show = [_][]const u8{ "commit abc", "Author: a", "", " msg", "", "diff --git a/x b/x" };
+ try testing.expect(looksLikeDiff(&show));
+ const prose = [_][]const u8{ "--- a heading", "+++ not a path", "text" };
+ try testing.expect(!looksLikeDiff(&prose));
+ const nothing = [_][]const u8{ "ls", "a.zig b.zig" };
+ try testing.expect(!looksLikeDiff(&nothing));
+}
+
+test "diff hunk headers: counts left out are one, a broken one is none" {
+ try testing.expectEqual(HunkHeader{ .old_start = 3, .old_count = 1, .new_start = 4, .new_count = 1 }, parseHunk("@@ -3 +4 @@").?);
+ try testing.expectEqual(HunkHeader{ .old_start = 1, .old_count = 0, .new_start = 0, .new_count = 0 }, parseHunk("@@ -1,0 +0,0 @@ fn x").?);
+ try testing.expect(parseHunk("@@ -a +1 @@") == null);
+ try testing.expect(parseHunk("@@ -1 +1") == null);
+ try testing.expect(parseHunk("@@@ -1 -1 +1 @@@") == null);
+}
+
+test "diff a terminal's wrapped row stays in its line" {
+ const lines = [_][]const u8{ "--- a/x.c", "+++ b/x.c", "@@ -1,2 +1,2 @@", "+int a = 1; /* a long", "comment */", " int b;", "-int c;" };
+ var w: Walk = .{ .continued_rows = true };
+ var kinds: [lines.len]Kind = undefined;
+ for (lines, 0..) |line, k| kinds[k] = w.step(line).kind;
+ try testing.expectEqualSlices(Kind, &.{ .old_path, .new_path, .hunk, .added, .added, .context, .removed }, &kinds);
+}