summaryrefslogtreecommitdiff
path: root/src/regexp.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-28 18:58:24 -0300
committerGabriel Schneider <[email protected]>2026-10-01 00:12:15 -0300
commitff867666106fa85bd91341beb9e8121b2f99bc5f (patch)
treea4a84b3f6f733673149fd2541cc9f70e61629fac /src/regexp.zig
parentab806dc38ddcd74958b999fe708ae9ea63609f72 (diff)
downloadpardes-ff867666106fa85bd91341beb9e8121b2f99bc5f.tar.gz
pardes-ff867666106fa85bd91341beb9e8121b2f99bc5f.zip
^ in a pattern with \n matches at every line start, and a ^ or $ that cannot work is refused
A pattern naming a newline runs over the whole text, where mvzr's ^ is only the text's start: /^def .*\n/ from #0 failed, and Edit ,x/^def .*\n/.../ silently did nothing. A leading ^ is now tried at each line start from the search's own; $ just before \n is dropped, as it changes nothing; any other ^ or $ in such a pattern is refused with why, since mvzr would read it as the text's ends. ^ inside an alternation holds only where a search starts, an mvzr limit, documented. Co-Authored-By: Claude Opus 5.5 <[email protected]>
Diffstat (limited to 'src/regexp.zig')
-rw-r--r--src/regexp.zig41
1 files changed, 40 insertions, 1 deletions
diff --git a/src/regexp.zig b/src/regexp.zig
index fe68585d..9c34ae2c 100644
--- a/src/regexp.zig
+++ b/src/regexp.zig
@@ -41,7 +41,12 @@ pub const Regex = struct {
/// patterns take 45-80 ms, and a Debug build is ten times slower.
pub const budget: u64 = if (builtin.mode == .Debug) 4_000_000 else 32_000_000;
- pub fn compile(pat: []const u8) error{Bad}!Regex {
+ pub const e_anchor = "in a pattern with \\n, ^ can only come first and $ only just before a \\n";
+
+ /// `Anchor`: a pattern that names a newline has `^` other than first,
+ /// or `$` other than just before a `\n`, which mvzr would read as the
+ /// ends of the whole text and so never match where sam would.
+ pub fn compile(pat: []const u8) error{ Bad, Anchor }!Regex {
if (pat.len == 0) return error.Bad;
var buf: [256]u8 = undefined;
var len: usize = 0;
@@ -68,6 +73,12 @@ pub const Regex = struct {
in_class = true;
} else if (c == '.' and spans) {
piece = "[^\\n]";
+ } else if (spans and c == '^' and i != 0) {
+ return error.Anchor;
+ } else if (spans and c == '$') {
+ // `x$\n` is `x\n`; any other `$` would be the text's end.
+ if (std.mem.startsWith(u8, pat[i + 1 ..], "\\n")) continue;
+ return error.Anchor;
}
if (!emit) continue;
if (len + piece.len > buf.len) return error.Bad;
@@ -91,6 +102,20 @@ pub const Regex = struct {
mvzr.steps_left = rx.steps;
mvzr.exhausted = false;
defer rx.steps = mvzr.steps_left;
+ // A `^` pattern that spans lines: mvzr's `^` is its haystack's start,
+ // so each line start from `from` on is tried as one.
+ // ponytail: a search per line start, each to `hi`; the step budget
+ // bounds it.
+ if (rx.spans and rx.bol) {
+ var s = if (from == 0 or text[from - 1] == '\n') from else (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1;
+ while (s <= last and s <= hi) {
+ const hit = rx.re.match(text[s..hi]);
+ if (mvzr.exhausted) return error.TooSlow;
+ if (hit) |m| if (m.start == 0) return .{ .start = s, .end = s + m.end };
+ s = (std.mem.indexOfScalarPos(u8, text[0..hi], s, '\n') orelse return null) + 1;
+ }
+ return null;
+ }
var start: usize = if (rx.spans) 0 else if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0;
var at = from - start;
// `^` cannot match in the middle of a line.
@@ -136,6 +161,20 @@ test "lines are haystacks: ^ and $ at each line, . never a newline, \\n spans li
try std.testing.expectEqual(c.end, m.end);
}
_ = try Regex.compile("a.*a\\nq");
+ // `^` at every line start and `$` before a newline, when a pattern
+ // spans lines; anywhere else they are refused, never silently wrong.
+ const defs = "x = 1\ndef a\n\ndef b\n";
+ var def = try Regex.compile("^def .*\\n");
+ const d = (try def.find(defs, 0, defs.len, defs.len)).?;
+ try std.testing.expectEqual(@as(usize, 6), d.start);
+ try std.testing.expectEqual(@as(usize, 12), d.end);
+ try std.testing.expectEqual(@as(usize, 13), (try def.find(defs, 7, defs.len, defs.len)).?.start);
+ var blank = try Regex.compile("^\\n");
+ try std.testing.expectEqual(@as(usize, 12), (try blank.find(defs, 0, defs.len, defs.len)).?.start);
+ var dollar = try Regex.compile("1$\\n");
+ try std.testing.expectEqual(@as(usize, 4), (try dollar.find(defs, 0, defs.len, defs.len)).?.start);
+ try std.testing.expectError(error.Anchor, Regex.compile("(^|\\n)def"));
+ try std.testing.expectError(error.Anchor, Regex.compile("a$\\nb$"));
var none = try Regex.compile("zzz");
try std.testing.expect(try none.find(text, 0, text.len, text.len) == null);
try std.testing.expectError(error.Bad, Regex.compile(""));