diff options
Diffstat (limited to 'src')
| -rw-r--r-- | src/ninep/addr.zig | 4 | ||||
| -rw-r--r-- | src/regexp.zig | 41 | ||||
| -rw-r--r-- | src/sam_edit.zig | 8 |
3 files changed, 49 insertions, 4 deletions
diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig index f57c62bd..1e47183b 100644 --- a/src/ninep/addr.zig +++ b/src/ninep/addr.zig @@ -222,8 +222,8 @@ pub const Addr = struct { a.err = "no previous regular expression"; return null; } - var rx = regexp_.Regex.compile(pat) catch { - a.err = e_regexp; + var rx = regexp_.Regex.compile(pat) catch |err| { + a.err = if (err == error.Anchor) regexp_.Regex.e_anchor else e_regexp; return null; }; const found = if (back) found: { diff --git a/src/regexp.zig b/src/regexp.zig index fe68585d..9c34ae2c 100644 --- a/src/regexp.zig +++ b/src/regexp.zig @@ -41,7 +41,12 @@ pub const Regex = struct { /// patterns take 45-80 ms, and a Debug build is ten times slower. pub const budget: u64 = if (builtin.mode == .Debug) 4_000_000 else 32_000_000; - pub fn compile(pat: []const u8) error{Bad}!Regex { + pub const e_anchor = "in a pattern with \\n, ^ can only come first and $ only just before a \\n"; + + /// `Anchor`: a pattern that names a newline has `^` other than first, + /// or `$` other than just before a `\n`, which mvzr would read as the + /// ends of the whole text and so never match where sam would. + pub fn compile(pat: []const u8) error{ Bad, Anchor }!Regex { if (pat.len == 0) return error.Bad; var buf: [256]u8 = undefined; var len: usize = 0; @@ -68,6 +73,12 @@ pub const Regex = struct { in_class = true; } else if (c == '.' and spans) { piece = "[^\\n]"; + } else if (spans and c == '^' and i != 0) { + return error.Anchor; + } else if (spans and c == '$') { + // `x$\n` is `x\n`; any other `$` would be the text's end. + if (std.mem.startsWith(u8, pat[i + 1 ..], "\\n")) continue; + return error.Anchor; } if (!emit) continue; if (len + piece.len > buf.len) return error.Bad; @@ -91,6 +102,20 @@ pub const Regex = struct { mvzr.steps_left = rx.steps; mvzr.exhausted = false; defer rx.steps = mvzr.steps_left; + // A `^` pattern that spans lines: mvzr's `^` is its haystack's start, + // so each line start from `from` on is tried as one. + // ponytail: a search per line start, each to `hi`; the step budget + // bounds it. + if (rx.spans and rx.bol) { + var s = if (from == 0 or text[from - 1] == '\n') from else (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1; + while (s <= last and s <= hi) { + const hit = rx.re.match(text[s..hi]); + if (mvzr.exhausted) return error.TooSlow; + if (hit) |m| if (m.start == 0) return .{ .start = s, .end = s + m.end }; + s = (std.mem.indexOfScalarPos(u8, text[0..hi], s, '\n') orelse return null) + 1; + } + return null; + } var start: usize = if (rx.spans) 0 else if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0; var at = from - start; // `^` cannot match in the middle of a line. @@ -136,6 +161,20 @@ test "lines are haystacks: ^ and $ at each line, . never a newline, \\n spans li try std.testing.expectEqual(c.end, m.end); } _ = try Regex.compile("a.*a\\nq"); + // `^` at every line start and `$` before a newline, when a pattern + // spans lines; anywhere else they are refused, never silently wrong. + const defs = "x = 1\ndef a\n\ndef b\n"; + var def = try Regex.compile("^def .*\\n"); + const d = (try def.find(defs, 0, defs.len, defs.len)).?; + try std.testing.expectEqual(@as(usize, 6), d.start); + try std.testing.expectEqual(@as(usize, 12), d.end); + try std.testing.expectEqual(@as(usize, 13), (try def.find(defs, 7, defs.len, defs.len)).?.start); + var blank = try Regex.compile("^\\n"); + try std.testing.expectEqual(@as(usize, 12), (try blank.find(defs, 0, defs.len, defs.len)).?.start); + var dollar = try Regex.compile("1$\\n"); + try std.testing.expectEqual(@as(usize, 4), (try dollar.find(defs, 0, defs.len, defs.len)).?.start); + try std.testing.expectError(error.Anchor, Regex.compile("(^|\\n)def")); + try std.testing.expectError(error.Anchor, Regex.compile("a$\\nb$")); var none = try Regex.compile("zzz"); try std.testing.expect(try none.find(text, 0, text.len, text.len) == null); try std.testing.expectError(error.Bad, Regex.compile("")); diff --git a/src/sam_edit.zig b/src/sam_edit.zig index c08c6cdd..40221628 100644 --- a/src/sam_edit.zig +++ b/src/sam_edit.zig @@ -380,7 +380,10 @@ const Exec = struct { } fn compile(ex: *Exec, pat: []const u8, c: u8) Failure!regexp.Regex { - return regexp.Regex.compile(pat) catch fail(ex.why, "bad regexp in {c} command", .{c}); + return regexp.Regex.compile(pat) catch |err| switch (err) { + error.Anchor => fail(ex.why, "{s}", .{regexp.Regex.e_anchor}), + error.Bad => fail(ex.why, "bad regexp in {c} command", .{c}), + }; } fn find(ex: *Exec, rx: *regexp.Regex, from: usize, hi: usize) Failure!?regexp.Regex.Match { @@ -637,6 +640,8 @@ test "sam's classic commands" { try expectEdit("one\ntwo\n", "1 m $", "two\none\n"); try expectEdit("one\ntwo\n", "1 t $", "one\ntwo\none\n"); try expectEdit("one\ntwo\nthree\n", "2 d", "one\nthree\n"); + // ^ at every line start in a pattern that spans lines (dogfood round 5). + try expectEdit("x = 1\ndef a\ndef b\n", ",x/^def .*\\n/a/X\\n/", "x = 1\ndef a\nX\ndef b\nX\n"); // pardes's line:col. try expectEdit("one\ntwo\n", "2:2 i/-/", "one\nt-wo\n"); } @@ -653,6 +658,7 @@ test "an Edit that fails halfway changes nothing, and says why in acme's words" .{ "w /tmp/x", "w is not supported in pardes" }, .{ ",s/(a)/\\1/", "no \\1: mvzr keeps no submatches" }, .{ "1 m 1,2", "move overlaps itself" }, + .{ ",x/(^|\\n)foo/d", "in a pattern with \\n, ^ can only come first and $ only just before a \\n" }, }) |c| { var why: Why = .{}; try std.testing.expectError(error.Edit, run(arena_state.allocator(), "foo a\nfoo\n", .{}, "t", c[0], &why)); |
