summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-28 11:12:10 -0300
committerGabriel Schneider <[email protected]>2026-10-01 00:12:14 -0300
commit5cf50930dee214fb641bb95bbf8639b8f76930d9 (patch)
treefe0855346e0c8d2b7e656cfa2cbc4a1abd9b9311 /src
parent3d8fc0733cd2d2c8a6209514aa1158cbd713be25 (diff)
downloadpardes-5cf50930dee214fb641bb95bbf8639b8f76930d9.tar.gz
pardes-5cf50930dee214fb641bb95bbf8639b8f76930d9.zip
Normal mode's s and S search as addr does, through one shared regexp.zig
The sam-style calling convention for mvzr (a line per haystack, . made [^\\n] only when a pattern names \\n) lived in addr.zig; normal mode's s and S called mvzr over the raw selection, where ^ meant the selection's start and . crossed lines. src/regexp.zig now holds the one Regex (compile, find) both call, and fs.zig and pardes.zig drop an unused mvzr import. hxdiff and hxparity stay at their known 17 and 6 mismatches, none of them regex cases. Co-Authored-By: Claude Opus 5.5 <[email protected]>
Diffstat (limited to 'src')
-rw-r--r--src/fs.zig1
-rw-r--r--src/ninep/addr.zig87
-rw-r--r--src/normal.zig53
-rw-r--r--src/pardes.zig2
-rw-r--r--src/regexp.zig104
5 files changed, 158 insertions, 89 deletions
diff --git a/src/fs.zig b/src/fs.zig
index 022a326a..0b620c9f 100644
--- a/src/fs.zig
+++ b/src/fs.zig
@@ -6,7 +6,6 @@ const config = @import("config.zig");
const lsp = @import("lsp/lsp.zig");
const ninep_io = @import("9p_io.zig");
const panes = @import("panes.zig");
-const mvzr = @import("mvzr");
const modal = @import("modal.zig");
const look = @import("look.zig");
pub const tree = @import("ninep/tree.zig");
diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig
index 95e699e3..347a2d11 100644
--- a/src/ninep/addr.zig
+++ b/src/ninep/addr.zig
@@ -1,7 +1,7 @@
//! The address language of a pane's `addr` file, acme's (editors/acme/
-//! addr.c), its regular expressions run by mvzr the way sam's run.
+//! addr.c), its regular expressions searched as sam searches (regexp.zig).
const std = @import("std");
-const mvzr = @import("mvzr");
+const regexp_ = @import("../regexp.zig");
const modal = @import("../modal.zig");
const pane_files = @import("pane.zig");
@@ -196,92 +196,27 @@ pub const Addr = struct {
a.err = "no previous regular expression";
return null;
}
- // sam searches the text as lines: `^` and `$` at any line's start
- // and end, and `.` never a newline. mvzr has no such mode (its `^`
- // and `$` are the haystack's ends, its `.` any byte), so each line
- // is its own haystack, and a pattern that names a newline (`\n`)
- // runs over the whole text with its `.`s made `[^\n]`.
- // ponytail: mvzr takes the first alternative that matches, not
- // sam's longest (`/gam|gamma/` finds `gam`); a search from the middle
- // of a line lets `^` match there unless the pattern starts with it;
- // across lines, `^`, `$` and `[^...]` keep mvzr's meaning. A regex
- // engine of sam's own would lift these; the user chose not to.
- const spans = std.mem.indexOf(u8, pat, "\\n") != null;
- var buf: [256]u8 = undefined;
- var len: usize = 0;
- var i: usize = 0;
- var in_class = false;
- while (i < pat.len) : (i += 1) {
- const c = pat[i];
- const piece: []const u8 = if (c == '\\') piece: {
- if (i + 1 >= pat.len) break :piece "";
- i += 1;
- break :piece pat[i - 1 .. i + 1];
- } else if (c == '.' and spans and !in_class) "[^\\n]" else pat[i .. i + 1];
- if (piece.len == 0 or len + piece.len > buf.len) {
- a.err = e_regexp;
- return null;
- }
- if (c == '[') in_class = true;
- if (c == ']') in_class = false;
- @memcpy(buf[len..][0..piece.len], piece);
- len += piece.len;
- }
- const re = mvzr.compile(buf[0..len]) orelse {
+ const rx = regexp_.Regex.compile(pat) orelse {
a.err = e_regexp;
return null;
};
- const anchored = pat[0] == '^';
- const Search = struct {
- /// The first match starting in `from..=last` that ends by `hi`.
- fn first(rx: *const mvzr.Regex, text: []const u8, from: usize, last: usize, hi: usize, whole: bool, bol: bool) ?Range {
- if (whole) {
- const m = rx.matchPos(from, text[0..hi]) orelse return null;
- return if (m.start <= last) .{ .q0 = clip(m.start), .q1 = clip(m.end) } else null;
- }
- var start = if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0;
- var at = from - start;
- // `^` cannot match in the middle of a line.
- if (bol and at > 0) {
- start = (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1;
- at = 0;
- }
- while (start <= hi and start <= last) {
- const end = std.mem.indexOfScalarPos(u8, text[0..hi], start, '\n') orelse hi;
- const line = text[start..end];
- // matchPos finds nothing at a line's very end, where
- // `$` or an empty pattern still match.
- const hit: ?[2]usize = if (at < line.len)
- (if (rx.matchPos(at, line)) |m| .{ m.start, m.end } else null)
- else if (at == line.len and rx.isMatch(line[at..])) .{ at, at } else null;
- if (hit) |h| {
- if (start + h[0] > last) return null;
- return .{ .q0 = clip(start + h[0]), .q1 = clip(start + h[1]) };
- }
- if (end == hi) return null;
- start = end + 1;
- at = 0;
- }
- return null;
- }
- };
- const found: ?Range = if (back) found: {
+ const found = if (back) found: {
var last: ?Range = null;
var before: ?Range = null;
var at: usize = 0;
while (at <= a.text.len) {
- const m = Search.first(&re, a.text, at, a.text.len, a.text.len, spans, anchored) orelse break;
- if (m.q1 <= r.q0) before = m;
- last = m;
- at = if (m.q1 > m.q0) m.q1 else m.q1 + 1;
+ const m = rx.find(a.text, at, a.text.len, a.text.len) orelse break;
+ const found_range: Range = .{ .q0 = clip(m.start), .q1 = clip(m.end) };
+ if (found_range.q1 <= r.q0) before = found_range;
+ last = found_range;
+ at = if (m.end > m.start) m.end else m.end + 1;
}
break :found before orelse last;
} else found: {
const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len;
const from = @min(@as(usize, r.q1), hi);
- if (Search.first(&re, a.text, from, hi, hi, spans, anchored)) |m| break :found m;
- if (a.lim != null or from == 0) break :found null;
- break :found Search.first(&re, a.text, 0, from - 1, hi, spans, anchored);
+ const m = rx.find(a.text, from, hi, hi) orelse (if (a.lim != null or from == 0) null else rx.find(a.text, 0, from - 1, hi)) orelse break :found null;
+ break :found Range{ .q0 = clip(m.start), .q1 = clip(m.end) };
};
return found orelse {
a.err = e_no_match;
diff --git a/src/normal.zig b/src/normal.zig
index ffbb49d5..8c5f232d 100644
--- a/src/normal.zig
+++ b/src/normal.zig
@@ -6,7 +6,7 @@ const tagline = @import("tagline.zig");
const exec = @import("exec.zig");
const look = @import("look.zig");
const std = @import("std");
-const mvzr = @import("mvzr");
+const regexp = @import("regexp.zig");
const modal = @import("modal.zig");
const panes = @import("panes.zig");
const edit = @import("edit.zig");
@@ -92,7 +92,7 @@ pub fn applySelRegex(p: *Pardes, pane: *Pane, t: *Text, pat: []const u8, split:
const snap = pane.sel_snap[0..pane.nsel_snap];
var out: [Text.max_selections]modal.Selection = undefined;
var m: usize = 0;
- if (pat.len > 0) if (mvzr.compile(pat)) |re| {
+ if (regexp.Regex.compile(pat)) |re| {
var hay_all = text;
if (for (pat) |c| {
if (std.ascii.isUpper(c)) break false;
@@ -105,17 +105,18 @@ pub fn applySelRegex(p: *Pardes, pane: *Pane, t: *Text, pat: []const u8, split:
const from = @min(r.anchor, r.head);
const to = @min(@max(r.anchor, r.head), text.len);
if (from >= to) continue;
- const hay = hay_all[from..to];
- var at: usize = 0;
+ // Searched as sam searches, the way addr does (regexp.zig): the
+ // text around the selection still says where its lines begin.
+ var at = from;
var piece = from; // split: where the next piece begins
- while (at < hay.len and m < Text.max_selections) {
- const hit_at = re.matchPos(at, hay) orelse break;
+ while (at < to and m < Text.max_selections) {
+ const hit_at = re.find(hay_all, at, to, to) orelse break;
if (split) {
- out[m] = .{ .anchor = piece, .head = from + hit_at.start };
+ out[m] = .{ .anchor = piece, .head = hit_at.start };
m += 1;
- piece = from + hit_at.end;
- } else if (from + hit_at.start != to) {
- out[m] = .{ .anchor = from + hit_at.start, .head = from + hit_at.end };
+ piece = hit_at.end;
+ } else if (hit_at.start != to) {
+ out[m] = .{ .anchor = hit_at.start, .head = hit_at.end };
m += 1;
}
// an empty match would otherwise never advance
@@ -126,7 +127,7 @@ pub fn applySelRegex(p: *Pardes, pane: *Pane, t: *Text, pat: []const u8, split:
m += 1;
}
}
- };
+ }
if (m == 0) {
// the text CAN move under an armed prompt (a tag chord runs a
// builtin), and these are raw offsets into the surface as it was
@@ -730,3 +731,33 @@ test "flat text movement and selection replay need no scratch rows" {
executeNormalAction(p, &pane.body, .match_bracket);
try std.testing.expectEqual(@as(i32, 8), pane.body.cur_col);
}
+
+test "s and S search a multi-line selection as addr does: ^ and $ per line, . within one" {
+ const gpa = std.testing.allocator;
+ const p = try Pardes.init(gpa, .{ .tty_only = true, .cols = 60, .rows = 12 });
+ defer p.deinit();
+ const text = "alpha x\nbeta alpha\nalphabet\n";
+ const pane = try p.setTestFile(text);
+ const Case = struct { pat: []const u8, split: bool, want: []const [2]usize };
+ for ([_]Case{
+ // ^ at each line's start, not only the selection's
+ .{ .pat = "^alpha", .split = false, .want = &.{ .{ 0, 5 }, .{ 19, 24 } } },
+ // $ at each line's end
+ .{ .pat = "alpha$", .split = false, .want = &.{.{ 13, 18 }} },
+ // . stops at the newline: one selection per line, not one for all
+ .{ .pat = "a.*", .split = false, .want = &.{ .{ 0, 7 }, .{ 11, 18 }, .{ 19, 27 } } },
+ // S splits on a line's end
+ .{ .pat = "$", .split = true, .want = &.{ .{ 0, 7 }, .{ 7, 18 }, .{ 18, 27 }, .{ 27, 28 } } },
+ }) |c| {
+ pane.sel_snap[0] = .{ .anchor = 0, .head = text.len };
+ pane.nsel_snap = 1;
+ applySelRegex(p, pane, &pane.body, c.pat, c.split);
+ var out: [Text.max_selections]modal.Selection = undefined;
+ const got = pane.body.ranges(text, 0, &out);
+ try std.testing.expectEqual(c.want.len, got.n);
+ for (c.want, out[0..got.n]) |w, r| {
+ try std.testing.expectEqual(w[0], @min(r.anchor, r.head));
+ try std.testing.expectEqual(w[1], @max(r.anchor, r.head));
+ }
+ }
+}
diff --git a/src/pardes.zig b/src/pardes.zig
index 9d336db3..05f6710a 100644
--- a/src/pardes.zig
+++ b/src/pardes.zig
@@ -2,7 +2,6 @@ const std = @import("std");
pub const layout = @import("layout.zig");
const uucode = @import("uucode");
const vaxis = @import("vaxis");
-const mvzr = @import("mvzr");
pub const modal = @import("modal.zig");
pub const look = @import("look.zig");
pub const filesystem = @import("fs.zig");
@@ -451,6 +450,7 @@ test {
_ = @import("look.zig");
_ = @import("mouse.zig");
_ = @import("normal.zig");
+ _ = @import("regexp.zig");
_ = @import("edit.zig");
_ = @import("Messages.zig");
_ = @import("selection_pipe.zig");
diff --git a/src/regexp.zig b/src/regexp.zig
new file mode 100644
index 00000000..665e51d5
--- /dev/null
+++ b/src/regexp.zig
@@ -0,0 +1,104 @@
+//! How pardes runs a regular expression over text: mvzr's, searched the way
+//! sam searches (editors/acme/regx.c). `addr` (src/ninep/addr.zig) and
+//! normal mode's `s` and `S` (src/normal.zig) both call it.
+const std = @import("std");
+const mvzr = @import("mvzr");
+
+/// A compiled pattern and how to run it.
+///
+/// sam searches the text as lines: `^` and `$` at any line's start and end,
+/// and `.` never a newline. mvzr has no such mode (its `^` and `$` are the
+/// haystack's ends, its `.` any byte), so each line is its own haystack, and
+/// a pattern that names a newline (`\n`) runs over the whole text with its
+/// `.`s made `[^\n]`.
+/// ponytail: mvzr takes the first alternative that matches, not sam's
+/// longest (`gam|gamma` finds `gam`); a search from the middle of a line lets
+/// `^` match there unless the pattern starts with it; across lines, `^`, `$`
+/// and `[^...]` keep mvzr's meaning. A regex engine of sam's own would lift
+/// these; the user chose not to have one.
+pub const Regex = struct {
+ re: mvzr.Regex,
+ /// The pattern names a newline: it runs over the whole text.
+ spans: bool,
+ /// The pattern starts with `^`: a search begun mid-line skips the line.
+ bol: bool,
+
+ pub fn compile(pat: []const u8) ?Regex {
+ if (pat.len == 0) return null;
+ const spans = std.mem.indexOf(u8, pat, "\\n") != null;
+ var buf: [256]u8 = undefined;
+ var len: usize = 0;
+ var i: usize = 0;
+ var in_class = false;
+ while (i < pat.len) : (i += 1) {
+ const c = pat[i];
+ const piece: []const u8 = if (c == '\\') piece: {
+ if (i + 1 >= pat.len) return null;
+ i += 1;
+ break :piece pat[i - 1 .. i + 1];
+ } else if (c == '.' and spans and !in_class) "[^\\n]" else pat[i .. i + 1];
+ if (len + piece.len > buf.len) return null;
+ if (c == '[') in_class = true;
+ if (c == ']') in_class = false;
+ @memcpy(buf[len..][0..piece.len], piece);
+ len += piece.len;
+ }
+ return .{ .re = mvzr.compile(buf[0..len]) orelse return null, .spans = spans, .bol = pat[0] == '^' };
+ }
+
+ /// The first match that starts in `from..=last` and ends by `hi`, as
+ /// offsets into `text`. The text before `from` still says where lines
+ /// begin.
+ pub fn find(rx: *const Regex, text: []const u8, from: usize, last: usize, hi: usize) ?struct { start: usize, end: usize } {
+ if (rx.spans) {
+ const m = rx.re.matchPos(from, text[0..hi]) orelse return null;
+ return if (m.start <= last) .{ .start = m.start, .end = m.end } else null;
+ }
+ var start = if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0;
+ var at = from - start;
+ // `^` cannot match in the middle of a line.
+ if (rx.bol and at > 0) {
+ start = (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1;
+ at = 0;
+ }
+ while (start <= hi and start <= last) {
+ const end = std.mem.indexOfScalarPos(u8, text[0..hi], start, '\n') orelse hi;
+ const line = text[start..end];
+ // matchPos finds nothing at a line's very end, where `$` or an
+ // empty match still can.
+ const hit: ?[2]usize = if (at < line.len)
+ (if (rx.re.matchPos(at, line)) |m| .{ m.start, m.end } else null)
+ else if (at == line.len and rx.re.isMatch(line[at..])) .{ at, at } else null;
+ if (hit) |h| {
+ if (start + h[0] > last) return null;
+ return .{ .start = start + h[0], .end = start + h[1] };
+ }
+ if (end == hi) return null;
+ start = end + 1;
+ at = 0;
+ }
+ return null;
+ }
+};
+
+test "lines are haystacks: ^ and $ at each line, . never a newline, \\n spans lines" {
+ const text = "alpha beta\nbeta gamma\ngamma\n";
+ const Case = struct { pat: []const u8, from: usize, start: usize, end: usize };
+ for ([_]Case{
+ .{ .pat = "^beta", .from = 0, .start = 11, .end = 15 },
+ .{ .pat = "beta$", .from = 0, .start = 6, .end = 10 },
+ .{ .pat = "a.*", .from = 0, .start = 0, .end = 10 },
+ .{ .pat = "a\\nbeta", .from = 0, .start = 9, .end = 15 },
+ .{ .pat = "t.\\nbeta", .from = 0, .start = 8, .end = 15 },
+ .{ .pat = "^", .from = 1, .start = 11, .end = 11 },
+ }) |c| {
+ const rx = Regex.compile(c.pat).?;
+ const m = rx.find(text, c.from, text.len, text.len).?;
+ try std.testing.expectEqual(c.start, m.start);
+ try std.testing.expectEqual(c.end, m.end);
+ }
+ try std.testing.expect(Regex.compile("a.*a\\nq") != null);
+ try std.testing.expect((Regex.compile("zzz").?).find(text, 0, text.len, text.len) == null);
+ try std.testing.expect(Regex.compile("") == null);
+ try std.testing.expect(Regex.compile("a\\") == null);
+}