summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
Diffstat (limited to 'src')
-rw-r--r--src/ninep/addr.zig2
-rw-r--r--src/regexp.zig159
-rw-r--r--src/sam_edit.zig2
3 files changed, 162 insertions, 1 deletions
diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig
index 005ad593..33a3f646 100644
--- a/src/ninep/addr.zig
+++ b/src/ninep/addr.zig
@@ -302,6 +302,8 @@ pub const Addr = struct {
a.err = switch (err) {
error.Anchor => regexp_.Regex.e_anchor,
error.TooLong => regexp_.Regex.e_long,
+ error.WideRange => regexp_.Regex.e_wide,
+ error.NegatedRunes => regexp_.Regex.e_negated,
error.Bad => e_regexp,
};
return null;
diff --git a/src/regexp.zig b/src/regexp.zig
index 695f2b87..ddb5eb28 100644
--- a/src/regexp.zig
+++ b/src/regexp.zig
@@ -54,6 +54,11 @@ pub const Regex = struct {
pub const e_long = std.fmt.comptimePrint("bad regular expression: longer than mvzr's {d} operations (about {d} pattern characters)", .{ max_ops, max_ops });
pub const e_anchor = "bad regular expression: in a pattern with \\n, ^ can only come first and $ only just before a \\n";
+ pub const e_wide = std.fmt.comptimePrint("bad regular expression: a range of runes wider than {d} in [...] is not supported", .{max_range});
+ pub const e_negated = "bad regular expression: a [^...] with non-ASCII runes is not supported (mvzr's classes hold bytes)";
+
+ /// The most runes a non-ASCII range in a class is spelled out as.
+ pub const max_range = 256;
/// `Anchor`: a pattern that names a newline has `^` other than first,
/// or `$` other than just before a `\n`, which mvzr would read as the
@@ -74,8 +79,14 @@ pub const Regex = struct {
return @intCast(n);
}
- pub fn compile(pat: []const u8) error{ Bad, Anchor, TooLong }!Regex {
+ pub const Error = error{ Bad, Anchor, TooLong, WideRange, NegatedRunes };
+
+ pub fn compile(pat: []const u8) Error!Regex {
if (pat.len == 0) return error.Bad;
+ // mvzr's classes hold bytes: `[éa-z]` is written `(é|[a-z])` for it.
+ // The limits apply to the pattern as rewritten.
+ var runes: [5 * max_ops + 2]u8 = undefined;
+ if (try runeClasses(pat, &runes)) |whole| return compile(whole);
// mvzr takes `^` only at its pattern's start, so `^def|^ ` (a `^`
// after a `|`) is written `^(def| )` for it: the same lines. A mix,
// `^a|b`, has no such spelling and is refused rather than wrong.
@@ -139,6 +150,126 @@ pub const Regex = struct {
};
}
+ /// `pat` with each class that holds a non-ASCII rune written as an
+ /// alternation of its runes and a class of the rest (`[éa-z]` is
+ /// `(é|[a-z])`), a range spelled out rune by rune; null when no class
+ /// holds one.
+ fn runeClasses(pat: []const u8, out: *[5 * max_ops + 2]u8) Error!?[]const u8 {
+ var w = std.Io.Writer.fixed(out);
+ var any = false;
+ var i: usize = 0;
+ while (i < pat.len) : (i += 1) {
+ const c = pat[i];
+ if (c == '\\') {
+ w.writeAll(pat[i..@min(i + 2, pat.len)]) catch return error.TooLong;
+ i += 1;
+ continue;
+ }
+ if (c != '[') {
+ w.writeByte(c) catch return error.TooLong;
+ continue;
+ }
+ // The class's end, as mvzr finds it: the first unescaped `]`.
+ var end = i + 1;
+ while (end < pat.len and pat[end] != ']') : (end += 1) {
+ if (pat[end] == '\\') end += 1;
+ }
+ if (end >= pat.len) return error.Bad;
+ const body = pat[i + 1 .. end];
+ if (for (body) |b| {
+ if (b >= 0x80) break false;
+ } else true) {
+ w.writeAll(pat[i .. end + 1]) catch return error.TooLong;
+ i = end;
+ continue;
+ }
+ if (body[0] == '^') return error.NegatedRunes;
+ any = true;
+ try runeClass(body, &w);
+ i = end;
+ }
+ return if (any) w.buffered() else null;
+ }
+
+ fn runeClass(body: []const u8, w: *std.Io.Writer) Error!void {
+ var ascii: [5 * max_ops]u8 = undefined;
+ var n: usize = 0;
+ w.writeByte('(') catch return error.TooLong;
+ var alts: usize = 0;
+ var j: usize = 0;
+ while (j < body.len) {
+ if (body[j] == '\\') {
+ const len: usize = if (j + 1 < body.len and body[j + 1] == 'x') 4 else 2;
+ if (j + len > body.len) return error.Bad;
+ // An escape at one end of a range whose other end is a rune.
+ if (j + len + 1 < body.len and body[j + len] == '-' and body[j + len + 1] >= 0x80) return error.Bad;
+ if (n + len > ascii.len) return error.TooLong;
+ @memcpy(ascii[n..][0..len], body[j..][0..len]);
+ n += len;
+ j += len;
+ continue;
+ }
+ const lo, const lo_len = try rune(body[j..]);
+ j += lo_len;
+ var hi = lo;
+ if (j + 1 < body.len and body[j] == '-') {
+ if (body[j + 1] == '\\' and lo >= 0x80) return error.Bad;
+ if (body[j + 1] != '\\') {
+ hi, const hi_len = try rune(body[j + 1 ..]);
+ j += 1 + hi_len;
+ }
+ }
+ if (hi < lo) return error.Bad;
+ if (hi - lo + 1 > max_range) return error.WideRange;
+ // Runes that differ only in their last byte go as one
+ // alternative, `\xc3[\xa0-\xbf]` for `[à-ÿ]`: 256 runes one a
+ // time would pass `max_ops`.
+ var run: [4]u8 = undefined;
+ var run_len: usize = 0;
+ var run_hi: u8 = 0;
+ var cp = lo;
+ while (cp <= hi + 1) : (cp += 1) {
+ var enc: [4]u8 = undefined;
+ const len = if (cp > hi) 0 else std.unicode.utf8Encode(cp, &enc) catch continue; // a surrogate
+ if (run_len > 0 and (len != run_len or enc[len - 1] != run_hi + 1 or
+ !std.mem.eql(u8, enc[0 .. len - 1], run[0 .. len - 1])))
+ {
+ if (alts > 0) w.writeByte('|') catch return error.TooLong;
+ w.writeAll(run[0 .. run_len - 1]) catch return error.TooLong;
+ if (run_hi == run[run_len - 1])
+ w.writeByte(run_hi) catch return error.TooLong
+ else
+ w.print("[\\x{x:0>2}-\\x{x:0>2}]", .{ run[run_len - 1], run_hi }) catch return error.TooLong;
+ alts += 1;
+ run_len = 0;
+ }
+ if (cp > hi) break;
+ if (cp < 0x80) {
+ // `\xHH` rather than the byte: `]`, `^`, `-` and `\`
+ // would otherwise mean something in the class.
+ if (n + 4 > ascii.len) return error.TooLong;
+ _ = std.fmt.bufPrint(ascii[n..][0..4], "\\x{x:0>2}", .{cp}) catch unreachable;
+ n += 4;
+ continue;
+ }
+ if (run_len == 0) {
+ run = enc;
+ run_len = len;
+ }
+ run_hi = enc[len - 1];
+ }
+ }
+ if (n > 0) w.print("|[{s}]", .{ascii[0..n]}) catch return error.TooLong;
+ w.writeByte(')') catch return error.TooLong;
+ }
+
+ /// The rune `s` opens with and its length in bytes.
+ fn rune(s: []const u8) error{Bad}!struct { u21, usize } {
+ const len = std.unicode.utf8ByteSequenceLength(s[0]) catch return error.Bad;
+ if (len > s.len) return error.Bad;
+ return .{ std.unicode.utf8Decode(s[0..len]) catch return error.Bad, len };
+ }
+
/// `^a|^b` as `^(a|b)` in `out`, when the pattern is an alternation at
/// its top level and every branch starts with `^`; null when it is not
/// one, or no branch does.
@@ -311,6 +442,32 @@ test "a pattern opening with a literal finds what a search from each line finds"
}
}
+test "a class with non-ASCII runes matches those runes: [éa-z], a range of them, and refuses a wide range or [^é]" {
+ const text = "1 é 2 b 3 ü 4 ñ\n";
+ var mixed = try Regex.compile("[éa-z]+");
+ try std.testing.expectEqual(@as(usize, 2), (try mixed.find(text, 0, text.len, text.len)).?.start);
+ const m = (try mixed.find(text, 5, text.len, text.len)).?;
+ try std.testing.expectEqualStrings("b", text[m.start..m.end]);
+ var range = try Regex.compile("3 [à-ÿ]");
+ const r = (try range.find(text, 0, text.len, text.len)).?;
+ try std.testing.expectEqualStrings("3 ü", text[r.start..r.end]);
+ // An ASCII member that means something in a class stays a member.
+ var odd = try Regex.compile("[ñ\\]^-]");
+ try std.testing.expectEqual(@as(usize, 16), (try odd.find(text, 0, text.len, text.len)).?.start);
+ var none = try Regex.compile("[ö]");
+ try std.testing.expect(try none.find(text, 0, text.len, text.len) == null);
+ try std.testing.expectError(error.WideRange, Regex.compile("[ā-ӿ]"));
+ try std.testing.expectError(error.NegatedRunes, Regex.compile("[^é]"));
+ try std.testing.expectError(error.Bad, Regex.compile("[é"));
+ try std.testing.expectError(error.Bad, Regex.compile("[ÿ-à]"));
+ // A full 256-rune range fits the limit as rewritten.
+ var wide = try Regex.compile("x[Ā-ǿ]");
+ try std.testing.expectEqual(@as(usize, 2), (try wide.find("ǿxǿ", 0, 5, 5)).?.start);
+ var kana = try Regex.compile("[ぁ-ゟ]"); // 3 bytes, across a last-byte wrap
+ try std.testing.expectEqual(@as(usize, 1), (try kana.find("aゞ", 0, 4, 4)).?.start);
+ try std.testing.expectError(error.WideRange, Regex.compile("[Ā-Ȁ]"));
+}
+
test "a pattern past 64 characters compiles, and one past the limit says so" {
var long = try Regex.compile("a" ** 200);
const text = "x" ++ "a" ** 200 ++ "\n";
diff --git a/src/sam_edit.zig b/src/sam_edit.zig
index ea1105bd..0f317a83 100644
--- a/src/sam_edit.zig
+++ b/src/sam_edit.zig
@@ -461,6 +461,8 @@ const Exec = struct {
return regexp.Regex.compile(pat) catch |err| switch (err) {
error.Anchor => fail(ex.why, "{s}", .{regexp.Regex.e_anchor}),
error.TooLong => fail(ex.why, "{s}", .{regexp.Regex.e_long}),
+ error.WideRange => fail(ex.why, "{s}", .{regexp.Regex.e_wide}),
+ error.NegatedRunes => fail(ex.why, "{s}", .{regexp.Regex.e_negated}),
error.Bad => fail(ex.why, "bad regexp in {c} command", .{c}),
};
}