diff options
Diffstat (limited to 'src/look.zig')
| -rw-r--r-- | src/look.zig | 144 |
1 files changed, 116 insertions, 28 deletions
diff --git a/src/look.zig b/src/look.zig index c4d04576..9df2b485 100644 --- a/src/look.zig +++ b/src/look.zig @@ -55,41 +55,125 @@ pub fn openLink(url: []const u8) void { _ = libc.waitpid(pid, null, 0); } -/// peel a trailing :LINE[:COL] suffix (both 1-based, 0 = absent): -/// main.zig:100 -> {main.zig, 100, 0} -/// main.zig:100:7 -> {main.zig, 100, 7} -/// main.zig:100: -> {main.zig, 100, 0} grep -n's trailing delimiter -pub fn parsePathLine(tok: []const u8) struct { path: []const u8, line: usize, col: usize } { +/// WHERE in a pane a look word points. A spot (`:LINE:COL`) — or a SPAN, when +/// the word carries a range (config.range_sep), which a look SELECTS instead +/// of merely parking on. Everything is 1-based and 0 means absent, so a bare +/// path is the all-zero Spot and `end_line == 0` is the question "is this a +/// range". +pub const Spot = struct { + line: usize = 0, + col: usize = 0, + end_line: usize = 0, + /// 0 with a live `end_line` is the whole-lines form: through the END of + /// end_line, newline included, which is what helix's `x` selects. + end_col: usize = 0, +}; + +/// digits at `i` and where they end; `end == i` means there were none. Four +/// numbers now come out of the same token, and spelling the scan four times +/// is how one of them ends up subtly different from the others. +fn num(tok: []const u8, i: usize) struct { v: usize, end: usize } { + var v: usize = 0; + var j = i; + while (j < tok.len and std.ascii.isDigit(tok[j])) : (j += 1) v = v * 10 + (tok[j] - '0'); + return .{ .v = v, .end = j }; +} + +/// peel a trailing :LINE[:COL] spot, or one of the three range spellings, off +/// a look word (config.line_col_sep / config.range_sep own both characters): +/// main.zig:100 -> line 100 +/// main.zig:100:7 -> line 100, col 7 +/// main.zig:100: -> line 100 grep -n's trailing delimiter +/// main.zig:100-104 -> lines 100..104 whole +/// main.zig:100:7-21 -> line 100, cols 7..21 +/// main.zig:100:7-104:3 -> line 100 col 7 .. line 104 col 3 +/// +/// A tail that does not parse leaves the token a plain PATH, which is the rule +/// that keeps the dash safe: `a-b`, `build-2:3` and `x:1-y` are all paths (the +/// last one goes back to hunting for a later ':' and finds none), because a +/// range needs a number on both sides of its dash. +pub fn parsePathLine(tok: []const u8) struct { path: []const u8, at: Spot } { var sep: usize = 0; while (sep < tok.len) : (sep += 1) { if (tok[sep] != config.line_col_sep) continue; - var j = sep + 1; - var line: usize = 0; - while (j < tok.len and std.ascii.isDigit(tok[j])) : (j += 1) line = line * 10 + (tok[j] - '0'); - if (j == sep + 1) continue; // no digits after ':' - if (j < tok.len and tok[j] != config.line_col_sep) continue; // junk after the number - var col: usize = 0; - if (j < tok.len) { - var k = j + 1; - while (k < tok.len and std.ascii.isDigit(tok[k])) : (k += 1) col = col * 10 + (tok[k] - '0'); - // digits, and nothing but a delimiter after them, or no column - if (k == j + 1 or (k < tok.len and tok[k] != config.line_col_sep)) col = 0; + const l = num(tok, sep + 1); + if (l.end == sep + 1) continue; // no digits after ':' + const path = tok[0..sep]; + var i = l.end; + // `:LINE-ENDLINE`: whole lines, no column anywhere in the form + if (i < tok.len and tok[i] == config.range_sep) { + const e = num(tok, i + 1); + if (e.end == i + 1) continue; // a dash with no number is not a range + if (e.end < tok.len and tok[e.end] != config.line_col_sep) continue; // junk after it + return .{ .path = path, .at = .{ .line = l.v, .end_line = e.v } }; } - return .{ .path = tok[0..sep], .line = line, .col = col }; + if (i < tok.len and tok[i] != config.line_col_sep) continue; // junk after the number + var at: Spot = .{ .line = l.v }; + if (i == tok.len) return .{ .path = path, .at = at }; + // `:COL`. A column that does not parse is dropped and the LINE still + // stands, which is how this has always read a half-mangled suffix. + const c = num(tok, i + 1); + if (c.end == i + 1) return .{ .path = path, .at = at }; + if (c.end < tok.len and tok[c.end] != config.line_col_sep and tok[c.end] != config.range_sep) + return .{ .path = path, .at = at }; + at.col = c.v; + i = c.end; + if (i == tok.len or tok[i] != config.range_sep) return .{ .path = path, .at = at }; + // `-ENDCOL` on this same line, unless a `:ENDCOL` follows — then that + // first number was the end LINE all along. One lookahead, and it is + // what lets the two-number and four-number forms share a spelling. + const e = num(tok, i + 1); + if (e.end == i + 1) return .{ .path = path, .at = at }; + at.end_line = at.line; + at.end_col = e.v; + if (e.end < tok.len and tok[e.end] == config.line_col_sep) { + const e2 = num(tok, e.end + 1); + if (e2.end > e.end + 1) { + at.end_line = at.end_col; + at.end_col = e2.v; + } + } + return .{ .path = path, .at = at }; + } + return .{ .path = tok, .at = .{} }; +} + +test "parsePathLine: spots, ranges, and the paths that merely look like them" { + const cases = [_]struct { tok: []const u8, path: []const u8, at: Spot }{ + .{ .tok = "main.zig", .path = "main.zig", .at = .{} }, + .{ .tok = "main.zig:100", .path = "main.zig", .at = .{ .line = 100 } }, + .{ .tok = "main.zig:100:", .path = "main.zig", .at = .{ .line = 100 } }, + .{ .tok = "main.zig:100:7", .path = "main.zig", .at = .{ .line = 100, .col = 7 } }, + .{ .tok = "main.zig:100-104", .path = "main.zig", .at = .{ .line = 100, .end_line = 104 } }, + .{ .tok = "main.zig:100:7-21", .path = "main.zig", .at = .{ .line = 100, .col = 7, .end_line = 100, .end_col = 21 } }, + .{ .tok = "main.zig:100:7-104:3", .path = "main.zig", .at = .{ .line = 100, .col = 7, .end_line = 104, .end_col = 3 } }, + // the dash cases that must stay ORDINARY PATHS + .{ .tok = "my-file.zig", .path = "my-file.zig", .at = .{} }, + .{ .tok = "my-file:10", .path = "my-file", .at = .{ .line = 10 } }, + .{ .tok = "x:1-y", .path = "x:1-y", .at = .{} }, + .{ .tok = "a-b-c", .path = "a-b-c", .at = .{} }, + .{ .tok = "2026-07-30", .path = "2026-07-30", .at = .{} }, + // a mangled tail still yields what parsed (unchanged behaviour) + .{ .tok = "main.zig:100x", .path = "main.zig:100x", .at = .{} }, + .{ .tok = "main.zig:100:7x", .path = "main.zig", .at = .{ .line = 100 } }, + }; + for (cases) |c| { + const got = parsePathLine(c.tok); + try std.testing.expectEqualStrings(c.path, got.path); + try std.testing.expectEqual(c.at, got.at); } - return .{ .path = tok, .line = 0, .col = 0 }; } pub const Target = union(enum) { none, dir: []const u8, // resolved absolute path, in caller's buf - file: struct { path: []const u8, line: usize, col: usize }, + file: struct { path: []const u8, at: Spot }, image: struct { path: []const u8 }, url: []const u8, /// `@p7:10:5` — pane 7, line 10, column 5 (0 = unspecified). The one /// target that names a live pane instead of a path, because terminals and /// output buffers have no file for a location to point at. - pane: struct { id: usize, line: usize, col: usize }, + pane: struct { id: usize, at: Spot }, }; pub fn isImagePath(path: []const u8) bool { @@ -114,7 +198,7 @@ pub fn resolve(word_raw: []const u8, cwd: []const u8, realbuf: *[4096]u8) Target for (word[config.pane_addr.len..]) |c| { if (!std.ascii.isDigit(c)) break; id = id * 10 + (c - '0'); - } else return .{ .pane = .{ .id = id, .line = pl.line, .col = pl.col } }; + } else return .{ .pane = .{ .id = id, .at = pl.at } }; } // a URL is a URL everywhere: no filesystem can answer it, so it leaves the @@ -134,12 +218,12 @@ pub fn resolve(word_raw: []const u8, cwd: []const u8, realbuf: *[4096]u8) Target const resolved = std.mem.span(rp); if (isDir(rp)) return .{ .dir = resolved }; if (isImagePath(resolved)) return .{ .image = .{ .path = resolved } }; - return .{ .file = .{ .path = resolved, .line = pl.line, .col = pl.col } }; + return .{ .file = .{ .path = resolved, .at = pl.at } }; } else { // web: tracked Zig sources resolve inside the build-generated, // read-only source filesystem. if (resolveEmbedded(word, cwd, realbuf)) |source| - return .{ .file = .{ .path = source.path, .line = pl.line, .col = pl.col } }; + return .{ .file = .{ .path = source.path, .at = pl.at } }; return .none; } } @@ -280,10 +364,12 @@ pub fn find(arena: std.mem.Allocator, dir: []const u8, pat: []const u8, out: *st const grep_max_bytes = 256 * 1024; const grep_max_files = 20_000; -/// every line of `text` holding `pat`, as `path:LINE:COL text` rows — the -/// shared half of grep(), and the shape every result row in pardes has: the -/// leading word is a look target, so n/N walk the hits. Returns the rows -/// written, at most `budget`. +/// every line of `text` holding `pat`, as `path:LINE:COL-ENDCOL text` rows — +/// the shared half of grep(), and the shape every result row in pardes has: +/// the leading word is a look target, so n/N walk the hits. The row names the +/// MATCH's span and not just its first cell, so stepping onto one selects the +/// text that matched (config.range_sep). Returns the rows written, at most +/// `budget`. fn grepText(arena: std.mem.Allocator, path: []const u8, text: []const u8, pat: []const u8, out: *std.ArrayList(u8), budget: usize) usize { var n: usize = 0; var line: usize = 0; @@ -298,7 +384,9 @@ fn grepText(arena: std.mem.Allocator, path: []const u8, text: []const u8, pat: [ const ln = std.mem.trimEnd(u8, raw, " \t\r"); var cut = @min(ln.len, 200); while (cut > 0 and cut < ln.len and ln[cut] & 0xc0 == 0x80) cut -= 1; - const row = std.fmt.allocPrint(arena, "{s}:{d}:{d} {s}\n", .{ path, line, at + 1, ln[0..cut] }) catch break; + const row = std.fmt.allocPrint(arena, "{s}:{d}:{d}{c}{d} {s}\n", .{ + path, line, at + 1, config.range_sep, at + pat.len, ln[0..cut], + }) catch break; out.appendSlice(arena, row) catch break; n += 1; } |
