summaryrefslogtreecommitdiff
path: root/src/sam_edit.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-29 03:01:30 -0300
committerGabriel Schneider <[email protected]>2026-10-01 00:12:15 -0300
commitbeea3aaa8d35c00145c78a746393d71b8b14e658 (patch)
tree85497b5c2097d595e3f86f33d269cf0c3336ef9f /src/sam_edit.zig
parentd0500fe7ecd6149ffafe6a04ff2b8206ff6c52df (diff)
downloadpardes-beea3aaa8d35c00145c78a746393d71b8b14e658.tar.gz
pardes-beea3aaa8d35c00145c78a746393d71b8b14e658.zip
Edit's x, y and s match whole runes, as addresses do
mvzr matches bytes, and only the addr path widened a match to runes, so ,x/./a/|/ on é split it into invalid UTF-8 and ,s/./X/g gave a combining sequence four X's. Edit's matches now go through the same rune alignment (and an empty match steps a rune, not a byte). runeStart no longer walks a stray continuation byte back onto the ASCII before it: an invalid byte is a place of its own. Co-Authored-By: Claude Opus 5.5 <[email protected]>
Diffstat (limited to 'src/sam_edit.zig')
-rw-r--r--src/sam_edit.zig35
1 files changed, 30 insertions, 5 deletions
diff --git a/src/sam_edit.zig b/src/sam_edit.zig
index 29dec726..95d59a9d 100644
--- a/src/sam_edit.zig
+++ b/src/sam_edit.zig
@@ -16,6 +16,13 @@
const std = @import("std");
const regexp = @import("regexp.zig");
const addr_lang = @import("ninep/addr.zig");
+const modal = @import("modal.zig");
+
+/// The next place after `p` a search may start: a rune on, and past the
+/// text's end once there, which ends a loop over `p <= end`.
+fn stepRune(text: []const u8, p: usize) usize {
+ return if (p >= text.len) p + 1 else modal.nextRune(text, p);
+}
pub const Range = addr_lang.Range;
@@ -418,9 +425,16 @@ const Exec = struct {
};
}
+ /// A match covers the runes it touches, as an address's does
+ /// (addr.zig): mvzr matches bytes, so `.` would take one byte of `é`
+ /// and x, y and s would split it. From `from` on only, so a match never
+ /// reaches back into what the last one took.
fn find(ex: *Exec, rx: *regexp.Regex, from: usize, hi: usize) Failure!?regexp.Regex.Match {
if (from > hi) return null;
- return rx.find(ex.text, from, hi, hi) catch fail(ex.why, "{s}", .{addr_lang.e_slow});
+ const m = (rx.find(ex.text, from, hi, hi) catch return fail(ex.why, "{s}", .{addr_lang.e_slow})) orelse return null;
+ var start = modal.runeStart(ex.text, m.start);
+ if (start < from) start = modal.runeEnd(ex.text, m.start);
+ return .{ .start = start, .end = @max(start, modal.runeEnd(ex.text, m.end)) };
}
/// ecmd.c:62, cmdexec: runs `c` from `dot`, answering the dot it leaves.
@@ -516,10 +530,10 @@ const Exec = struct {
const m = (try ex.find(&rx, p1, r.q1)) orelse break;
if (m.start == m.end) {
if (op != null and m.start == op.?) {
- p1 += 1;
+ p1 = stepRune(ex.text, p1);
continue;
}
- p1 = m.end + 1;
+ p1 = stepRune(ex.text, m.end);
} else p1 = m.end;
op = m.end;
n -= 1;
@@ -560,10 +574,10 @@ const Exec = struct {
if (try ex.find(&rx, p, r.q1)) |m| {
if (m.start == m.end) {
if (op != null and m.start == op.?) {
- p += 1;
+ p = stepRune(ex.text, p);
continue;
}
- p = m.end + 1;
+ p = stepRune(ex.text, m.end);
} else p = m.end;
tr = if (xy) .{ .q0 = clip(m.start), .q1 = clip(m.end) } else .{ .q0 = clip(op.?), .q1 = clip(m.start) };
sel_end = m.end;
@@ -723,3 +737,14 @@ test "the dot an Edit leaves selects what a change put where it stood" {
try std.testing.expectEqual(Range{ .q0 = 2, .q1 = 5 }, moveDot(.{ .q0 = 2, .q1 = 2 }, &ops));
try std.testing.expectEqual(Range{ .q0 = 6, .q1 = 7 }, moveDot(.{ .q0 = 5, .q1 = 6 }, &ops));
}
+
+test "x, y and s match whole runes: a byte pattern never splits one" {
+ // é is c3 a9: `.` takes it whole, as it takes a lone combining mark.
+ try expectEdit("\u{e9}", ",x/./a/|/", "\u{e9}|");
+ try expectEdit("e\u{301}t", ",s/./X/g", "XXX");
+ try expectEdit("a\u{e9}b", ",y/\u{e9}/c/-/", "-\u{e9}-");
+ try expectEdit("\u{e9}\u{e9}", ",x/\\xC3/c/E/", "EE");
+ // An invalid byte is a place of its own.
+ try expectEdit("a\x81b", ",x/./a/|/", "a|\x81|b|");
+ try expectEdit("\u{e9}", ",s/$/!/", "\u{e9}!");
+}