summaryrefslogtreecommitdiff
path: root/src/look.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/look.zig')
-rw-r--r--src/look.zig144
1 files changed, 116 insertions, 28 deletions
diff --git a/src/look.zig b/src/look.zig
index c4d04576..9df2b485 100644
--- a/src/look.zig
+++ b/src/look.zig
@@ -55,41 +55,125 @@ pub fn openLink(url: []const u8) void {
_ = libc.waitpid(pid, null, 0);
}
-/// peel a trailing :LINE[:COL] suffix (both 1-based, 0 = absent):
-/// main.zig:100 -> {main.zig, 100, 0}
-/// main.zig:100:7 -> {main.zig, 100, 7}
-/// main.zig:100: -> {main.zig, 100, 0} grep -n's trailing delimiter
-pub fn parsePathLine(tok: []const u8) struct { path: []const u8, line: usize, col: usize } {
+/// WHERE in a pane a look word points. A spot (`:LINE:COL`) — or a SPAN, when
+/// the word carries a range (config.range_sep), which a look SELECTS instead
+/// of merely parking on. Everything is 1-based and 0 means absent, so a bare
+/// path is the all-zero Spot and `end_line == 0` is the question "is this a
+/// range".
+pub const Spot = struct {
+ line: usize = 0,
+ col: usize = 0,
+ end_line: usize = 0,
+ /// 0 with a live `end_line` is the whole-lines form: through the END of
+ /// end_line, newline included, which is what helix's `x` selects.
+ end_col: usize = 0,
+};
+
+/// digits at `i` and where they end; `end == i` means there were none. Four
+/// numbers now come out of the same token, and spelling the scan four times
+/// is how one of them ends up subtly different from the others.
+fn num(tok: []const u8, i: usize) struct { v: usize, end: usize } {
+ var v: usize = 0;
+ var j = i;
+ while (j < tok.len and std.ascii.isDigit(tok[j])) : (j += 1) v = v * 10 + (tok[j] - '0');
+ return .{ .v = v, .end = j };
+}
+
+/// peel a trailing :LINE[:COL] spot, or one of the three range spellings, off
+/// a look word (config.line_col_sep / config.range_sep own both characters):
+/// main.zig:100 -> line 100
+/// main.zig:100:7 -> line 100, col 7
+/// main.zig:100: -> line 100 grep -n's trailing delimiter
+/// main.zig:100-104 -> lines 100..104 whole
+/// main.zig:100:7-21 -> line 100, cols 7..21
+/// main.zig:100:7-104:3 -> line 100 col 7 .. line 104 col 3
+///
+/// A tail that does not parse leaves the token a plain PATH, which is the rule
+/// that keeps the dash safe: `a-b`, `build-2:3` and `x:1-y` are all paths (the
+/// last one goes back to hunting for a later ':' and finds none), because a
+/// range needs a number on both sides of its dash.
+pub fn parsePathLine(tok: []const u8) struct { path: []const u8, at: Spot } {
var sep: usize = 0;
while (sep < tok.len) : (sep += 1) {
if (tok[sep] != config.line_col_sep) continue;
- var j = sep + 1;
- var line: usize = 0;
- while (j < tok.len and std.ascii.isDigit(tok[j])) : (j += 1) line = line * 10 + (tok[j] - '0');
- if (j == sep + 1) continue; // no digits after ':'
- if (j < tok.len and tok[j] != config.line_col_sep) continue; // junk after the number
- var col: usize = 0;
- if (j < tok.len) {
- var k = j + 1;
- while (k < tok.len and std.ascii.isDigit(tok[k])) : (k += 1) col = col * 10 + (tok[k] - '0');
- // digits, and nothing but a delimiter after them, or no column
- if (k == j + 1 or (k < tok.len and tok[k] != config.line_col_sep)) col = 0;
+ const l = num(tok, sep + 1);
+ if (l.end == sep + 1) continue; // no digits after ':'
+ const path = tok[0..sep];
+ var i = l.end;
+ // `:LINE-ENDLINE`: whole lines, no column anywhere in the form
+ if (i < tok.len and tok[i] == config.range_sep) {
+ const e = num(tok, i + 1);
+ if (e.end == i + 1) continue; // a dash with no number is not a range
+ if (e.end < tok.len and tok[e.end] != config.line_col_sep) continue; // junk after it
+ return .{ .path = path, .at = .{ .line = l.v, .end_line = e.v } };
}
- return .{ .path = tok[0..sep], .line = line, .col = col };
+ if (i < tok.len and tok[i] != config.line_col_sep) continue; // junk after the number
+ var at: Spot = .{ .line = l.v };
+ if (i == tok.len) return .{ .path = path, .at = at };
+ // `:COL`. A column that does not parse is dropped and the LINE still
+ // stands, which is how this has always read a half-mangled suffix.
+ const c = num(tok, i + 1);
+ if (c.end == i + 1) return .{ .path = path, .at = at };
+ if (c.end < tok.len and tok[c.end] != config.line_col_sep and tok[c.end] != config.range_sep)
+ return .{ .path = path, .at = at };
+ at.col = c.v;
+ i = c.end;
+ if (i == tok.len or tok[i] != config.range_sep) return .{ .path = path, .at = at };
+ // `-ENDCOL` on this same line, unless a `:ENDCOL` follows — then that
+ // first number was the end LINE all along. One lookahead, and it is
+ // what lets the two-number and four-number forms share a spelling.
+ const e = num(tok, i + 1);
+ if (e.end == i + 1) return .{ .path = path, .at = at };
+ at.end_line = at.line;
+ at.end_col = e.v;
+ if (e.end < tok.len and tok[e.end] == config.line_col_sep) {
+ const e2 = num(tok, e.end + 1);
+ if (e2.end > e.end + 1) {
+ at.end_line = at.end_col;
+ at.end_col = e2.v;
+ }
+ }
+ return .{ .path = path, .at = at };
+ }
+ return .{ .path = tok, .at = .{} };
+}
+
+test "parsePathLine: spots, ranges, and the paths that merely look like them" {
+ const cases = [_]struct { tok: []const u8, path: []const u8, at: Spot }{
+ .{ .tok = "main.zig", .path = "main.zig", .at = .{} },
+ .{ .tok = "main.zig:100", .path = "main.zig", .at = .{ .line = 100 } },
+ .{ .tok = "main.zig:100:", .path = "main.zig", .at = .{ .line = 100 } },
+ .{ .tok = "main.zig:100:7", .path = "main.zig", .at = .{ .line = 100, .col = 7 } },
+ .{ .tok = "main.zig:100-104", .path = "main.zig", .at = .{ .line = 100, .end_line = 104 } },
+ .{ .tok = "main.zig:100:7-21", .path = "main.zig", .at = .{ .line = 100, .col = 7, .end_line = 100, .end_col = 21 } },
+ .{ .tok = "main.zig:100:7-104:3", .path = "main.zig", .at = .{ .line = 100, .col = 7, .end_line = 104, .end_col = 3 } },
+ // the dash cases that must stay ORDINARY PATHS
+ .{ .tok = "my-file.zig", .path = "my-file.zig", .at = .{} },
+ .{ .tok = "my-file:10", .path = "my-file", .at = .{ .line = 10 } },
+ .{ .tok = "x:1-y", .path = "x:1-y", .at = .{} },
+ .{ .tok = "a-b-c", .path = "a-b-c", .at = .{} },
+ .{ .tok = "2026-07-30", .path = "2026-07-30", .at = .{} },
+ // a mangled tail still yields what parsed (unchanged behaviour)
+ .{ .tok = "main.zig:100x", .path = "main.zig:100x", .at = .{} },
+ .{ .tok = "main.zig:100:7x", .path = "main.zig", .at = .{ .line = 100 } },
+ };
+ for (cases) |c| {
+ const got = parsePathLine(c.tok);
+ try std.testing.expectEqualStrings(c.path, got.path);
+ try std.testing.expectEqual(c.at, got.at);
}
- return .{ .path = tok, .line = 0, .col = 0 };
}
pub const Target = union(enum) {
none,
dir: []const u8, // resolved absolute path, in caller's buf
- file: struct { path: []const u8, line: usize, col: usize },
+ file: struct { path: []const u8, at: Spot },
image: struct { path: []const u8 },
url: []const u8,
/// `@p7:10:5` — pane 7, line 10, column 5 (0 = unspecified). The one
/// target that names a live pane instead of a path, because terminals and
/// output buffers have no file for a location to point at.
- pane: struct { id: usize, line: usize, col: usize },
+ pane: struct { id: usize, at: Spot },
};
pub fn isImagePath(path: []const u8) bool {
@@ -114,7 +198,7 @@ pub fn resolve(word_raw: []const u8, cwd: []const u8, realbuf: *[4096]u8) Target
for (word[config.pane_addr.len..]) |c| {
if (!std.ascii.isDigit(c)) break;
id = id * 10 + (c - '0');
- } else return .{ .pane = .{ .id = id, .line = pl.line, .col = pl.col } };
+ } else return .{ .pane = .{ .id = id, .at = pl.at } };
}
// a URL is a URL everywhere: no filesystem can answer it, so it leaves the
@@ -134,12 +218,12 @@ pub fn resolve(word_raw: []const u8, cwd: []const u8, realbuf: *[4096]u8) Target
const resolved = std.mem.span(rp);
if (isDir(rp)) return .{ .dir = resolved };
if (isImagePath(resolved)) return .{ .image = .{ .path = resolved } };
- return .{ .file = .{ .path = resolved, .line = pl.line, .col = pl.col } };
+ return .{ .file = .{ .path = resolved, .at = pl.at } };
} else {
// web: tracked Zig sources resolve inside the build-generated,
// read-only source filesystem.
if (resolveEmbedded(word, cwd, realbuf)) |source|
- return .{ .file = .{ .path = source.path, .line = pl.line, .col = pl.col } };
+ return .{ .file = .{ .path = source.path, .at = pl.at } };
return .none;
}
}
@@ -280,10 +364,12 @@ pub fn find(arena: std.mem.Allocator, dir: []const u8, pat: []const u8, out: *st
const grep_max_bytes = 256 * 1024;
const grep_max_files = 20_000;
-/// every line of `text` holding `pat`, as `path:LINE:COL text` rows — the
-/// shared half of grep(), and the shape every result row in pardes has: the
-/// leading word is a look target, so n/N walk the hits. Returns the rows
-/// written, at most `budget`.
+/// every line of `text` holding `pat`, as `path:LINE:COL-ENDCOL text` rows —
+/// the shared half of grep(), and the shape every result row in pardes has:
+/// the leading word is a look target, so n/N walk the hits. The row names the
+/// MATCH's span and not just its first cell, so stepping onto one selects the
+/// text that matched (config.range_sep). Returns the rows written, at most
+/// `budget`.
fn grepText(arena: std.mem.Allocator, path: []const u8, text: []const u8, pat: []const u8, out: *std.ArrayList(u8), budget: usize) usize {
var n: usize = 0;
var line: usize = 0;
@@ -298,7 +384,9 @@ fn grepText(arena: std.mem.Allocator, path: []const u8, text: []const u8, pat: [
const ln = std.mem.trimEnd(u8, raw, " \t\r");
var cut = @min(ln.len, 200);
while (cut > 0 and cut < ln.len and ln[cut] & 0xc0 == 0x80) cut -= 1;
- const row = std.fmt.allocPrint(arena, "{s}:{d}:{d} {s}\n", .{ path, line, at + 1, ln[0..cut] }) catch break;
+ const row = std.fmt.allocPrint(arena, "{s}:{d}:{d}{c}{d} {s}\n", .{
+ path, line, at + 1, config.range_sep, at + pat.len, ln[0..cut],
+ }) catch break;
out.appendSlice(arena, row) catch break;
n += 1;
}