summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--docs/fs.md5
-rw-r--r--src/ninep/addr.zig6
-rw-r--r--src/regexp.zig34
-rw-r--r--src/sam_edit.zig1
4 files changed, 37 insertions, 9 deletions
diff --git a/docs/fs.md b/docs/fs.md
index 7b1e47bf..9ca5f936 100644
--- a/docs/fs.md
+++ b/docs/fs.md
@@ -713,7 +713,10 @@ whole text: never a search that silently finds nothing. An alternation
whose every branch starts with `^` (`^def|^ `) finds a line that starts
either way (it is taken as `^(def| )`); one that mixes anchored and
unanchored branches (`^def|x`) is refused, `bad regular expression`, since
-mvzr keeps `^` only first. An expression is
+mvzr keeps `^` only first. A pattern may be up to 512 of mvzr's operations,
+about 512 characters (mvzr's own is 64; pardes builds it with more); a
+longer one is refused, `bad regular expression: longer than mvzr's 512
+operations`. An expression is
evaluated from the current address, the range last written to `addr` (or
left by the last `data` write, just past it), as acme evaluates it from
`w->addr` (xfid.c:446): `.` is that address, not the selection (`dot` is
diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig
index f60c07b1..c052a665 100644
--- a/src/ninep/addr.zig
+++ b/src/ninep/addr.zig
@@ -299,7 +299,11 @@ pub const Addr = struct {
return null;
}
var rx = regexp_.Regex.compile(pat) catch |err| {
- a.err = if (err == error.Anchor) regexp_.Regex.e_anchor else e_regexp;
+ a.err = switch (err) {
+ error.Anchor => regexp_.Regex.e_anchor,
+ error.TooLong => regexp_.Regex.e_long,
+ error.Bad => e_regexp,
+ };
return null;
};
// sam's nextmatch (editors/sam/address.c:97-119): an empty match
diff --git a/src/regexp.zig b/src/regexp.zig
index ca396ff1..dd2e7157 100644
--- a/src/regexp.zig
+++ b/src/regexp.zig
@@ -27,8 +27,14 @@ const mvzr = @import("mvzr");
/// enough line (`\s*(\w+)\s*=` over 20 KB of letters) runs out of budget. A
/// regex engine of sam's own would lift these; the user chose not to have
/// one.
+/// mvzr's own `Regex` holds 64 operations, some 64 pattern characters; a
+/// search pattern is often longer. Past these a pattern is refused naming
+/// the limit (`e_long`).
+pub const max_ops = 512;
+const Compiled = mvzr.SizedRegex(max_ops, 64);
+
pub const Regex = struct {
- re: mvzr.Regex,
+ re: Compiled,
/// The pattern names a newline: it runs over the whole text.
spans: bool,
/// The pattern starts with `^`: a search begun mid-line skips the line.
@@ -41,19 +47,21 @@ pub const Regex = struct {
/// patterns take 45-80 ms, and a Debug build is ten times slower.
pub const budget: u64 = if (builtin.mode == .Debug) 4_000_000 else 32_000_000;
+ pub const e_long = std.fmt.comptimePrint("bad regular expression: longer than mvzr's {d} operations (about {d} pattern characters)", .{ max_ops, max_ops });
pub const e_anchor = "bad regular expression: in a pattern with \\n, ^ can only come first and $ only just before a \\n";
/// `Anchor`: a pattern that names a newline has `^` other than first,
/// or `$` other than just before a `\n`, which mvzr would read as the
/// ends of the whole text and so never match where sam would.
- pub fn compile(pat: []const u8) error{ Bad, Anchor }!Regex {
+ pub fn compile(pat: []const u8) error{ Bad, Anchor, TooLong }!Regex {
if (pat.len == 0) return error.Bad;
// mvzr takes `^` only at its pattern's start, so `^def|^ ` (a `^`
// after a `|`) is written `^(def| )` for it: the same lines. A mix,
// `^a|b`, has no such spelling and is refused rather than wrong.
- var joined: [258]u8 = undefined;
+ var joined: [5 * max_ops + 2]u8 = undefined;
if (try anchoredAlternation(pat, &joined)) |whole| return compile(whole);
- var buf: [256]u8 = undefined;
+ // `.` may become `[^\n]`: five bytes for one.
+ var buf: [5 * max_ops]u8 = undefined;
var len: usize = 0;
var spans = false;
// Twice over the pattern: the first pass learns whether it names a
@@ -90,13 +98,17 @@ pub const Regex = struct {
return error.Anchor;
}
if (!emit) continue;
- if (len + piece.len > buf.len) return error.Bad;
+ if (len + piece.len > buf.len) return error.TooLong;
@memcpy(buf[len..][0..piece.len], piece);
len += piece.len;
}
}
return .{
- .re = mvzr.compile(buf[0..len]) orelse return error.Bad,
+ .re = Compiled.compile(buf[0..len]) orelse {
+ // Too long, or malformed: told apart by trying it with room.
+ if (mvzr.SizedRegex(4 * max_ops, 256).compile(buf[0..len]) != null) return error.TooLong;
+ return error.Bad;
+ },
.spans = spans,
.bol = pat[0] == '^',
};
@@ -105,7 +117,7 @@ pub const Regex = struct {
/// `^a|^b` as `^(a|b)` in `out`, when the pattern is an alternation at
/// its top level and every branch starts with `^`; null when it is not
/// one, or no branch does.
- fn anchoredAlternation(pat: []const u8, out: *[258]u8) error{Bad}!?[]const u8 {
+ fn anchoredAlternation(pat: []const u8, out: *[5 * max_ops + 2]u8) error{Bad}!?[]const u8 {
var bars: [16]usize = undefined;
var n: usize = 0;
var depth: usize = 0;
@@ -240,6 +252,14 @@ test "lines are haystacks: ^ and $ at each line, . never a newline, \\n spans li
try std.testing.expectError(error.Bad, Regex.compile("a\\"));
}
+test "a pattern past 64 characters compiles, and one past the limit says so" {
+ var long = try Regex.compile("a" ** 200);
+ const text = "x" ++ "a" ** 200 ++ "\n";
+ try std.testing.expectEqual(@as(usize, 1), (try long.find(text, 0, text.len, text.len)).?.start);
+ try std.testing.expectError(error.TooLong, Regex.compile("a" ** (max_ops + 8)));
+ try std.testing.expectError(error.Bad, Regex.compile("a[b"));
+}
+
test "a ^ after | anchors that branch: ^def|^ finds a line that starts either way, and a mix is refused" {
const text = "x def\n a\ndef b\n";
var both = try Regex.compile("^def|^ ");
diff --git a/src/sam_edit.zig b/src/sam_edit.zig
index a88728aa..113ad14f 100644
--- a/src/sam_edit.zig
+++ b/src/sam_edit.zig
@@ -445,6 +445,7 @@ const Exec = struct {
fn compile(ex: *Exec, pat: []const u8, c: u8) Failure!regexp.Regex {
return regexp.Regex.compile(pat) catch |err| switch (err) {
error.Anchor => fail(ex.why, "{s}", .{regexp.Regex.e_anchor}),
+ error.TooLong => fail(ex.why, "{s}", .{regexp.Regex.e_long}),
error.Bad => fail(ex.why, "bad regexp in {c} command", .{c}),
};
}