diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-29 17:46:11 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-10-01 00:12:17 -0300 |
| commit | e3e605bd33fc5be56cb7ec842449718bc5337a75 (patch) | |
| tree | 85a9d70524ecb3358bc3f92225851c6abd249a10 /src | |
| parent | 290584fc1127d017603c7c98c9aab27c0523a2cf (diff) | |
| download | pardes-e3e605bd33fc5be56cb7ec842449718bc5337a75.tar.gz pardes-e3e605bd33fc5be56cb7ec842449718bc5337a75.zip | |
A class with non-ASCII runes matches them: [éa-z] is taken as (é|[a-z]), a range up to 256 runes spelled out
mvzr's classes hold bytes, and it refused a multibyte member as a bad expression. Rewriting the class before compiling, as ^a|^b already is, needs no mvzr patch; runes that differ only in their last byte go as one alternative (\xc3[\xa0-\xbf] for [à-ÿ]) so a 256-rune range fits the 512-operation limit, which counts the pattern as rewritten. A wider range and [^é] are refused, naming why.
Co-Authored-By: Claude Opus 5.5 <[email protected]>
Diffstat (limited to 'src')
| -rw-r--r-- | src/ninep/addr.zig | 2 | ||||
| -rw-r--r-- | src/regexp.zig | 159 | ||||
| -rw-r--r-- | src/sam_edit.zig | 2 |
3 files changed, 162 insertions, 1 deletions
diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig index 005ad593..33a3f646 100644 --- a/src/ninep/addr.zig +++ b/src/ninep/addr.zig @@ -302,6 +302,8 @@ pub const Addr = struct { a.err = switch (err) { error.Anchor => regexp_.Regex.e_anchor, error.TooLong => regexp_.Regex.e_long, + error.WideRange => regexp_.Regex.e_wide, + error.NegatedRunes => regexp_.Regex.e_negated, error.Bad => e_regexp, }; return null; diff --git a/src/regexp.zig b/src/regexp.zig index 695f2b87..ddb5eb28 100644 --- a/src/regexp.zig +++ b/src/regexp.zig @@ -54,6 +54,11 @@ pub const Regex = struct { pub const e_long = std.fmt.comptimePrint("bad regular expression: longer than mvzr's {d} operations (about {d} pattern characters)", .{ max_ops, max_ops }); pub const e_anchor = "bad regular expression: in a pattern with \\n, ^ can only come first and $ only just before a \\n"; + pub const e_wide = std.fmt.comptimePrint("bad regular expression: a range of runes wider than {d} in [...] is not supported", .{max_range}); + pub const e_negated = "bad regular expression: a [^...] with non-ASCII runes is not supported (mvzr's classes hold bytes)"; + + /// The most runes a non-ASCII range in a class is spelled out as. + pub const max_range = 256; /// `Anchor`: a pattern that names a newline has `^` other than first, /// or `$` other than just before a `\n`, which mvzr would read as the @@ -74,8 +79,14 @@ pub const Regex = struct { return @intCast(n); } - pub fn compile(pat: []const u8) error{ Bad, Anchor, TooLong }!Regex { + pub const Error = error{ Bad, Anchor, TooLong, WideRange, NegatedRunes }; + + pub fn compile(pat: []const u8) Error!Regex { if (pat.len == 0) return error.Bad; + // mvzr's classes hold bytes: `[éa-z]` is written `(é|[a-z])` for it. + // The limits apply to the pattern as rewritten. + var runes: [5 * max_ops + 2]u8 = undefined; + if (try runeClasses(pat, &runes)) |whole| return compile(whole); // mvzr takes `^` only at its pattern's start, so `^def|^ ` (a `^` // after a `|`) is written `^(def| )` for it: the same lines. A mix, // `^a|b`, has no such spelling and is refused rather than wrong. @@ -139,6 +150,126 @@ pub const Regex = struct { }; } + /// `pat` with each class that holds a non-ASCII rune written as an + /// alternation of its runes and a class of the rest (`[éa-z]` is + /// `(é|[a-z])`), a range spelled out rune by rune; null when no class + /// holds one. + fn runeClasses(pat: []const u8, out: *[5 * max_ops + 2]u8) Error!?[]const u8 { + var w = std.Io.Writer.fixed(out); + var any = false; + var i: usize = 0; + while (i < pat.len) : (i += 1) { + const c = pat[i]; + if (c == '\\') { + w.writeAll(pat[i..@min(i + 2, pat.len)]) catch return error.TooLong; + i += 1; + continue; + } + if (c != '[') { + w.writeByte(c) catch return error.TooLong; + continue; + } + // The class's end, as mvzr finds it: the first unescaped `]`. + var end = i + 1; + while (end < pat.len and pat[end] != ']') : (end += 1) { + if (pat[end] == '\\') end += 1; + } + if (end >= pat.len) return error.Bad; + const body = pat[i + 1 .. end]; + if (for (body) |b| { + if (b >= 0x80) break false; + } else true) { + w.writeAll(pat[i .. end + 1]) catch return error.TooLong; + i = end; + continue; + } + if (body[0] == '^') return error.NegatedRunes; + any = true; + try runeClass(body, &w); + i = end; + } + return if (any) w.buffered() else null; + } + + fn runeClass(body: []const u8, w: *std.Io.Writer) Error!void { + var ascii: [5 * max_ops]u8 = undefined; + var n: usize = 0; + w.writeByte('(') catch return error.TooLong; + var alts: usize = 0; + var j: usize = 0; + while (j < body.len) { + if (body[j] == '\\') { + const len: usize = if (j + 1 < body.len and body[j + 1] == 'x') 4 else 2; + if (j + len > body.len) return error.Bad; + // An escape at one end of a range whose other end is a rune. + if (j + len + 1 < body.len and body[j + len] == '-' and body[j + len + 1] >= 0x80) return error.Bad; + if (n + len > ascii.len) return error.TooLong; + @memcpy(ascii[n..][0..len], body[j..][0..len]); + n += len; + j += len; + continue; + } + const lo, const lo_len = try rune(body[j..]); + j += lo_len; + var hi = lo; + if (j + 1 < body.len and body[j] == '-') { + if (body[j + 1] == '\\' and lo >= 0x80) return error.Bad; + if (body[j + 1] != '\\') { + hi, const hi_len = try rune(body[j + 1 ..]); + j += 1 + hi_len; + } + } + if (hi < lo) return error.Bad; + if (hi - lo + 1 > max_range) return error.WideRange; + // Runes that differ only in their last byte go as one + // alternative, `\xc3[\xa0-\xbf]` for `[à-ÿ]`: 256 runes one a + // time would pass `max_ops`. + var run: [4]u8 = undefined; + var run_len: usize = 0; + var run_hi: u8 = 0; + var cp = lo; + while (cp <= hi + 1) : (cp += 1) { + var enc: [4]u8 = undefined; + const len = if (cp > hi) 0 else std.unicode.utf8Encode(cp, &enc) catch continue; // a surrogate + if (run_len > 0 and (len != run_len or enc[len - 1] != run_hi + 1 or + !std.mem.eql(u8, enc[0 .. len - 1], run[0 .. len - 1]))) + { + if (alts > 0) w.writeByte('|') catch return error.TooLong; + w.writeAll(run[0 .. run_len - 1]) catch return error.TooLong; + if (run_hi == run[run_len - 1]) + w.writeByte(run_hi) catch return error.TooLong + else + w.print("[\\x{x:0>2}-\\x{x:0>2}]", .{ run[run_len - 1], run_hi }) catch return error.TooLong; + alts += 1; + run_len = 0; + } + if (cp > hi) break; + if (cp < 0x80) { + // `\xHH` rather than the byte: `]`, `^`, `-` and `\` + // would otherwise mean something in the class. + if (n + 4 > ascii.len) return error.TooLong; + _ = std.fmt.bufPrint(ascii[n..][0..4], "\\x{x:0>2}", .{cp}) catch unreachable; + n += 4; + continue; + } + if (run_len == 0) { + run = enc; + run_len = len; + } + run_hi = enc[len - 1]; + } + } + if (n > 0) w.print("|[{s}]", .{ascii[0..n]}) catch return error.TooLong; + w.writeByte(')') catch return error.TooLong; + } + + /// The rune `s` opens with and its length in bytes. + fn rune(s: []const u8) error{Bad}!struct { u21, usize } { + const len = std.unicode.utf8ByteSequenceLength(s[0]) catch return error.Bad; + if (len > s.len) return error.Bad; + return .{ std.unicode.utf8Decode(s[0..len]) catch return error.Bad, len }; + } + /// `^a|^b` as `^(a|b)` in `out`, when the pattern is an alternation at /// its top level and every branch starts with `^`; null when it is not /// one, or no branch does. @@ -311,6 +442,32 @@ test "a pattern opening with a literal finds what a search from each line finds" } } +test "a class with non-ASCII runes matches those runes: [éa-z], a range of them, and refuses a wide range or [^é]" { + const text = "1 é 2 b 3 ü 4 ñ\n"; + var mixed = try Regex.compile("[éa-z]+"); + try std.testing.expectEqual(@as(usize, 2), (try mixed.find(text, 0, text.len, text.len)).?.start); + const m = (try mixed.find(text, 5, text.len, text.len)).?; + try std.testing.expectEqualStrings("b", text[m.start..m.end]); + var range = try Regex.compile("3 [à-ÿ]"); + const r = (try range.find(text, 0, text.len, text.len)).?; + try std.testing.expectEqualStrings("3 ü", text[r.start..r.end]); + // An ASCII member that means something in a class stays a member. + var odd = try Regex.compile("[ñ\\]^-]"); + try std.testing.expectEqual(@as(usize, 16), (try odd.find(text, 0, text.len, text.len)).?.start); + var none = try Regex.compile("[ö]"); + try std.testing.expect(try none.find(text, 0, text.len, text.len) == null); + try std.testing.expectError(error.WideRange, Regex.compile("[ā-ӿ]")); + try std.testing.expectError(error.NegatedRunes, Regex.compile("[^é]")); + try std.testing.expectError(error.Bad, Regex.compile("[é")); + try std.testing.expectError(error.Bad, Regex.compile("[ÿ-à]")); + // A full 256-rune range fits the limit as rewritten. + var wide = try Regex.compile("x[Ā-ǿ]"); + try std.testing.expectEqual(@as(usize, 2), (try wide.find("ǿxǿ", 0, 5, 5)).?.start); + var kana = try Regex.compile("[ぁ-ゟ]"); // 3 bytes, across a last-byte wrap + try std.testing.expectEqual(@as(usize, 1), (try kana.find("aゞ", 0, 4, 4)).?.start); + try std.testing.expectError(error.WideRange, Regex.compile("[Ā-Ȁ]")); +} + test "a pattern past 64 characters compiles, and one past the limit says so" { var long = try Regex.compile("a" ** 200); const text = "x" ++ "a" ** 200 ++ "\n"; diff --git a/src/sam_edit.zig b/src/sam_edit.zig index ea1105bd..0f317a83 100644 --- a/src/sam_edit.zig +++ b/src/sam_edit.zig @@ -461,6 +461,8 @@ const Exec = struct { return regexp.Regex.compile(pat) catch |err| switch (err) { error.Anchor => fail(ex.why, "{s}", .{regexp.Regex.e_anchor}), error.TooLong => fail(ex.why, "{s}", .{regexp.Regex.e_long}), + error.WideRange => fail(ex.why, "{s}", .{regexp.Regex.e_wide}), + error.NegatedRunes => fail(ex.why, "{s}", .{regexp.Regex.e_negated}), error.Bad => fail(ex.why, "bad regexp in {c} command", .{c}), }; } |
