diff options
| -rw-r--r-- | docs/fs.md | 2 | ||||
| -rw-r--r-- | docs/helix-keys.md | 6 | ||||
| -rw-r--r-- | src/fs.zig | 1 | ||||
| -rw-r--r-- | src/ninep/addr.zig | 87 | ||||
| -rw-r--r-- | src/normal.zig | 53 | ||||
| -rw-r--r-- | src/pardes.zig | 2 | ||||
| -rw-r--r-- | src/regexp.zig | 104 |
7 files changed, 165 insertions, 90 deletions
@@ -274,6 +274,8 @@ takes the longest (`/gam|gamma/` finds `gam`); in a search begun in the middle of a line, `^` inside an alternation can match there; and in a pattern that spans lines, `^`, `$` and `[^...]` keep mvzr's own meaning. pardes has no regex engine of its own on purpose; these are its limits. +Normal mode's `s` and `S` search a selection the same way (src/regexp.zig +is the one place both call), so `^` there also means a line's start. An address that does not evaluate fails the write with why: `bad address syntax`, `no match for regexp`, `address out of range` or `bad regular diff --git a/docs/helix-keys.md b/docs/helix-keys.md index f46f5234..0f847405 100644 --- a/docs/helix-keys.md +++ b/docs/helix-keys.md @@ -245,7 +245,11 @@ re-runs the final pattern down the same path, so a submit can never disagree with what is on screen. Pressed in a tag, column tag or workspace tag they select in that tag's own text, and so does `|` pipe it; `/` searches the body wherever it is pressed (`Pane.prompt_for`). Engine: **mvzr** (`build.zig.zon`), a bytecode VM that -compiles a runtime pattern with no allocator. +compiles a runtime pattern with no allocator, called the way sam searches +(`src/regexp.zig`, shared with a pane's `addr` file): each line is its own +haystack, so `^` and `$` match at every line's start and end and `.` never +crosses a newline, and a pattern naming `\n` runs over the whole selection. +Where helix's `^` is only the selection's start, this is sam's. | Key | Behavior | Notes | Status | | --- | --- | --- | --- | @@ -6,7 +6,6 @@ const config = @import("config.zig"); const lsp = @import("lsp/lsp.zig"); const ninep_io = @import("9p_io.zig"); const panes = @import("panes.zig"); -const mvzr = @import("mvzr"); const modal = @import("modal.zig"); const look = @import("look.zig"); pub const tree = @import("ninep/tree.zig"); diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig index 95e699e3..347a2d11 100644 --- a/src/ninep/addr.zig +++ b/src/ninep/addr.zig @@ -1,7 +1,7 @@ //! The address language of a pane's `addr` file, acme's (editors/acme/ -//! addr.c), its regular expressions run by mvzr the way sam's run. +//! addr.c), its regular expressions searched as sam searches (regexp.zig). const std = @import("std"); -const mvzr = @import("mvzr"); +const regexp_ = @import("../regexp.zig"); const modal = @import("../modal.zig"); const pane_files = @import("pane.zig"); @@ -196,92 +196,27 @@ pub const Addr = struct { a.err = "no previous regular expression"; return null; } - // sam searches the text as lines: `^` and `$` at any line's start - // and end, and `.` never a newline. mvzr has no such mode (its `^` - // and `$` are the haystack's ends, its `.` any byte), so each line - // is its own haystack, and a pattern that names a newline (`\n`) - // runs over the whole text with its `.`s made `[^\n]`. - // ponytail: mvzr takes the first alternative that matches, not - // sam's longest (`/gam|gamma/` finds `gam`); a search from the middle - // of a line lets `^` match there unless the pattern starts with it; - // across lines, `^`, `$` and `[^...]` keep mvzr's meaning. A regex - // engine of sam's own would lift these; the user chose not to. - const spans = std.mem.indexOf(u8, pat, "\\n") != null; - var buf: [256]u8 = undefined; - var len: usize = 0; - var i: usize = 0; - var in_class = false; - while (i < pat.len) : (i += 1) { - const c = pat[i]; - const piece: []const u8 = if (c == '\\') piece: { - if (i + 1 >= pat.len) break :piece ""; - i += 1; - break :piece pat[i - 1 .. i + 1]; - } else if (c == '.' and spans and !in_class) "[^\\n]" else pat[i .. i + 1]; - if (piece.len == 0 or len + piece.len > buf.len) { - a.err = e_regexp; - return null; - } - if (c == '[') in_class = true; - if (c == ']') in_class = false; - @memcpy(buf[len..][0..piece.len], piece); - len += piece.len; - } - const re = mvzr.compile(buf[0..len]) orelse { + const rx = regexp_.Regex.compile(pat) orelse { a.err = e_regexp; return null; }; - const anchored = pat[0] == '^'; - const Search = struct { - /// The first match starting in `from..=last` that ends by `hi`. - fn first(rx: *const mvzr.Regex, text: []const u8, from: usize, last: usize, hi: usize, whole: bool, bol: bool) ?Range { - if (whole) { - const m = rx.matchPos(from, text[0..hi]) orelse return null; - return if (m.start <= last) .{ .q0 = clip(m.start), .q1 = clip(m.end) } else null; - } - var start = if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0; - var at = from - start; - // `^` cannot match in the middle of a line. - if (bol and at > 0) { - start = (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1; - at = 0; - } - while (start <= hi and start <= last) { - const end = std.mem.indexOfScalarPos(u8, text[0..hi], start, '\n') orelse hi; - const line = text[start..end]; - // matchPos finds nothing at a line's very end, where - // `$` or an empty pattern still match. - const hit: ?[2]usize = if (at < line.len) - (if (rx.matchPos(at, line)) |m| .{ m.start, m.end } else null) - else if (at == line.len and rx.isMatch(line[at..])) .{ at, at } else null; - if (hit) |h| { - if (start + h[0] > last) return null; - return .{ .q0 = clip(start + h[0]), .q1 = clip(start + h[1]) }; - } - if (end == hi) return null; - start = end + 1; - at = 0; - } - return null; - } - }; - const found: ?Range = if (back) found: { + const found = if (back) found: { var last: ?Range = null; var before: ?Range = null; var at: usize = 0; while (at <= a.text.len) { - const m = Search.first(&re, a.text, at, a.text.len, a.text.len, spans, anchored) orelse break; - if (m.q1 <= r.q0) before = m; - last = m; - at = if (m.q1 > m.q0) m.q1 else m.q1 + 1; + const m = rx.find(a.text, at, a.text.len, a.text.len) orelse break; + const found_range: Range = .{ .q0 = clip(m.start), .q1 = clip(m.end) }; + if (found_range.q1 <= r.q0) before = found_range; + last = found_range; + at = if (m.end > m.start) m.end else m.end + 1; } break :found before orelse last; } else found: { const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len; const from = @min(@as(usize, r.q1), hi); - if (Search.first(&re, a.text, from, hi, hi, spans, anchored)) |m| break :found m; - if (a.lim != null or from == 0) break :found null; - break :found Search.first(&re, a.text, 0, from - 1, hi, spans, anchored); + const m = rx.find(a.text, from, hi, hi) orelse (if (a.lim != null or from == 0) null else rx.find(a.text, 0, from - 1, hi)) orelse break :found null; + break :found Range{ .q0 = clip(m.start), .q1 = clip(m.end) }; }; return found orelse { a.err = e_no_match; diff --git a/src/normal.zig b/src/normal.zig index ffbb49d5..8c5f232d 100644 --- a/src/normal.zig +++ b/src/normal.zig @@ -6,7 +6,7 @@ const tagline = @import("tagline.zig"); const exec = @import("exec.zig"); const look = @import("look.zig"); const std = @import("std"); -const mvzr = @import("mvzr"); +const regexp = @import("regexp.zig"); const modal = @import("modal.zig"); const panes = @import("panes.zig"); const edit = @import("edit.zig"); @@ -92,7 +92,7 @@ pub fn applySelRegex(p: *Pardes, pane: *Pane, t: *Text, pat: []const u8, split: const snap = pane.sel_snap[0..pane.nsel_snap]; var out: [Text.max_selections]modal.Selection = undefined; var m: usize = 0; - if (pat.len > 0) if (mvzr.compile(pat)) |re| { + if (regexp.Regex.compile(pat)) |re| { var hay_all = text; if (for (pat) |c| { if (std.ascii.isUpper(c)) break false; @@ -105,17 +105,18 @@ pub fn applySelRegex(p: *Pardes, pane: *Pane, t: *Text, pat: []const u8, split: const from = @min(r.anchor, r.head); const to = @min(@max(r.anchor, r.head), text.len); if (from >= to) continue; - const hay = hay_all[from..to]; - var at: usize = 0; + // Searched as sam searches, the way addr does (regexp.zig): the + // text around the selection still says where its lines begin. + var at = from; var piece = from; // split: where the next piece begins - while (at < hay.len and m < Text.max_selections) { - const hit_at = re.matchPos(at, hay) orelse break; + while (at < to and m < Text.max_selections) { + const hit_at = re.find(hay_all, at, to, to) orelse break; if (split) { - out[m] = .{ .anchor = piece, .head = from + hit_at.start }; + out[m] = .{ .anchor = piece, .head = hit_at.start }; m += 1; - piece = from + hit_at.end; - } else if (from + hit_at.start != to) { - out[m] = .{ .anchor = from + hit_at.start, .head = from + hit_at.end }; + piece = hit_at.end; + } else if (hit_at.start != to) { + out[m] = .{ .anchor = hit_at.start, .head = hit_at.end }; m += 1; } // an empty match would otherwise never advance @@ -126,7 +127,7 @@ pub fn applySelRegex(p: *Pardes, pane: *Pane, t: *Text, pat: []const u8, split: m += 1; } } - }; + } if (m == 0) { // the text CAN move under an armed prompt (a tag chord runs a // builtin), and these are raw offsets into the surface as it was @@ -730,3 +731,33 @@ test "flat text movement and selection replay need no scratch rows" { executeNormalAction(p, &pane.body, .match_bracket); try std.testing.expectEqual(@as(i32, 8), pane.body.cur_col); } + +test "s and S search a multi-line selection as addr does: ^ and $ per line, . within one" { + const gpa = std.testing.allocator; + const p = try Pardes.init(gpa, .{ .tty_only = true, .cols = 60, .rows = 12 }); + defer p.deinit(); + const text = "alpha x\nbeta alpha\nalphabet\n"; + const pane = try p.setTestFile(text); + const Case = struct { pat: []const u8, split: bool, want: []const [2]usize }; + for ([_]Case{ + // ^ at each line's start, not only the selection's + .{ .pat = "^alpha", .split = false, .want = &.{ .{ 0, 5 }, .{ 19, 24 } } }, + // $ at each line's end + .{ .pat = "alpha$", .split = false, .want = &.{.{ 13, 18 }} }, + // . stops at the newline: one selection per line, not one for all + .{ .pat = "a.*", .split = false, .want = &.{ .{ 0, 7 }, .{ 11, 18 }, .{ 19, 27 } } }, + // S splits on a line's end + .{ .pat = "$", .split = true, .want = &.{ .{ 0, 7 }, .{ 7, 18 }, .{ 18, 27 }, .{ 27, 28 } } }, + }) |c| { + pane.sel_snap[0] = .{ .anchor = 0, .head = text.len }; + pane.nsel_snap = 1; + applySelRegex(p, pane, &pane.body, c.pat, c.split); + var out: [Text.max_selections]modal.Selection = undefined; + const got = pane.body.ranges(text, 0, &out); + try std.testing.expectEqual(c.want.len, got.n); + for (c.want, out[0..got.n]) |w, r| { + try std.testing.expectEqual(w[0], @min(r.anchor, r.head)); + try std.testing.expectEqual(w[1], @max(r.anchor, r.head)); + } + } +} diff --git a/src/pardes.zig b/src/pardes.zig index 9d336db3..05f6710a 100644 --- a/src/pardes.zig +++ b/src/pardes.zig @@ -2,7 +2,6 @@ const std = @import("std"); pub const layout = @import("layout.zig"); const uucode = @import("uucode"); const vaxis = @import("vaxis"); -const mvzr = @import("mvzr"); pub const modal = @import("modal.zig"); pub const look = @import("look.zig"); pub const filesystem = @import("fs.zig"); @@ -451,6 +450,7 @@ test { _ = @import("look.zig"); _ = @import("mouse.zig"); _ = @import("normal.zig"); + _ = @import("regexp.zig"); _ = @import("edit.zig"); _ = @import("Messages.zig"); _ = @import("selection_pipe.zig"); diff --git a/src/regexp.zig b/src/regexp.zig new file mode 100644 index 00000000..665e51d5 --- /dev/null +++ b/src/regexp.zig @@ -0,0 +1,104 @@ +//! How pardes runs a regular expression over text: mvzr's, searched the way +//! sam searches (editors/acme/regx.c). `addr` (src/ninep/addr.zig) and +//! normal mode's `s` and `S` (src/normal.zig) both call it. +const std = @import("std"); +const mvzr = @import("mvzr"); + +/// A compiled pattern and how to run it. +/// +/// sam searches the text as lines: `^` and `$` at any line's start and end, +/// and `.` never a newline. mvzr has no such mode (its `^` and `$` are the +/// haystack's ends, its `.` any byte), so each line is its own haystack, and +/// a pattern that names a newline (`\n`) runs over the whole text with its +/// `.`s made `[^\n]`. +/// ponytail: mvzr takes the first alternative that matches, not sam's +/// longest (`gam|gamma` finds `gam`); a search from the middle of a line lets +/// `^` match there unless the pattern starts with it; across lines, `^`, `$` +/// and `[^...]` keep mvzr's meaning. A regex engine of sam's own would lift +/// these; the user chose not to have one. +pub const Regex = struct { + re: mvzr.Regex, + /// The pattern names a newline: it runs over the whole text. + spans: bool, + /// The pattern starts with `^`: a search begun mid-line skips the line. + bol: bool, + + pub fn compile(pat: []const u8) ?Regex { + if (pat.len == 0) return null; + const spans = std.mem.indexOf(u8, pat, "\\n") != null; + var buf: [256]u8 = undefined; + var len: usize = 0; + var i: usize = 0; + var in_class = false; + while (i < pat.len) : (i += 1) { + const c = pat[i]; + const piece: []const u8 = if (c == '\\') piece: { + if (i + 1 >= pat.len) return null; + i += 1; + break :piece pat[i - 1 .. i + 1]; + } else if (c == '.' and spans and !in_class) "[^\\n]" else pat[i .. i + 1]; + if (len + piece.len > buf.len) return null; + if (c == '[') in_class = true; + if (c == ']') in_class = false; + @memcpy(buf[len..][0..piece.len], piece); + len += piece.len; + } + return .{ .re = mvzr.compile(buf[0..len]) orelse return null, .spans = spans, .bol = pat[0] == '^' }; + } + + /// The first match that starts in `from..=last` and ends by `hi`, as + /// offsets into `text`. The text before `from` still says where lines + /// begin. + pub fn find(rx: *const Regex, text: []const u8, from: usize, last: usize, hi: usize) ?struct { start: usize, end: usize } { + if (rx.spans) { + const m = rx.re.matchPos(from, text[0..hi]) orelse return null; + return if (m.start <= last) .{ .start = m.start, .end = m.end } else null; + } + var start = if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0; + var at = from - start; + // `^` cannot match in the middle of a line. + if (rx.bol and at > 0) { + start = (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1; + at = 0; + } + while (start <= hi and start <= last) { + const end = std.mem.indexOfScalarPos(u8, text[0..hi], start, '\n') orelse hi; + const line = text[start..end]; + // matchPos finds nothing at a line's very end, where `$` or an + // empty match still can. + const hit: ?[2]usize = if (at < line.len) + (if (rx.re.matchPos(at, line)) |m| .{ m.start, m.end } else null) + else if (at == line.len and rx.re.isMatch(line[at..])) .{ at, at } else null; + if (hit) |h| { + if (start + h[0] > last) return null; + return .{ .start = start + h[0], .end = start + h[1] }; + } + if (end == hi) return null; + start = end + 1; + at = 0; + } + return null; + } +}; + +test "lines are haystacks: ^ and $ at each line, . never a newline, \\n spans lines" { + const text = "alpha beta\nbeta gamma\ngamma\n"; + const Case = struct { pat: []const u8, from: usize, start: usize, end: usize }; + for ([_]Case{ + .{ .pat = "^beta", .from = 0, .start = 11, .end = 15 }, + .{ .pat = "beta$", .from = 0, .start = 6, .end = 10 }, + .{ .pat = "a.*", .from = 0, .start = 0, .end = 10 }, + .{ .pat = "a\\nbeta", .from = 0, .start = 9, .end = 15 }, + .{ .pat = "t.\\nbeta", .from = 0, .start = 8, .end = 15 }, + .{ .pat = "^", .from = 1, .start = 11, .end = 11 }, + }) |c| { + const rx = Regex.compile(c.pat).?; + const m = rx.find(text, c.from, text.len, text.len).?; + try std.testing.expectEqual(c.start, m.start); + try std.testing.expectEqual(c.end, m.end); + } + try std.testing.expect(Regex.compile("a.*a\\nq") != null); + try std.testing.expect((Regex.compile("zzz").?).find(text, 0, text.len, text.len) == null); + try std.testing.expect(Regex.compile("") == null); + try std.testing.expect(Regex.compile("a\\") == null); +} |
