summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--docs/fs.md2
-rw-r--r--src/modal.zig22
-rw-r--r--src/sam_edit.zig35
3 files changed, 46 insertions, 13 deletions
diff --git a/docs/fs.md b/docs/fs.md
index c95a45db..428cc84c 100644
--- a/docs/fs.md
+++ b/docs/fs.md
@@ -664,7 +664,7 @@ inside a multibyte rune and never widened to a grapheme cluster: a `#n`
inside a rune snaps back to its start, a `line:col` likewise, a search's
match covers the runes it touches, and a copy of addr to dot keeps its
runes, so a lone combining mark or the `\r` of a CRLF is addressable on its
-own; `addr` reads back the snapped offset. (How the terminal draws such a
+own (an Edit's `x`, `y` and `s` match the same way: `.` is one rune); `addr` reads back the snapped offset. (How the terminal draws such a
text, a cluster to a cell, is apart from this.) A click in a
tag gives offsets into the whole tag as `tag` reads it, the path first.) An
open freezes the ring's text the way `/screen` freezes a frame: reads walk it
diff --git a/src/modal.zig b/src/modal.zig
index dd7d2b57..11eda7b9 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -1327,18 +1327,26 @@ pub fn lineEndOffset(text: []const u8, line: usize) usize {
/// acme's do, not grapheme clusters: a lone combining mark or a `\r` is a
/// place of its own, whatever a terminal draws in one cell.
pub fn runeStart(text: []const u8, off: usize) usize {
- var o = @min(off, text.len);
+ const o = @min(off, text.len);
+ if (o == text.len or text[o] & 0xC0 != 0x80) return o;
+ var s = o;
var steps: u8 = 0;
- while (o > 0 and o < text.len and text[o] & 0xC0 == 0x80 and steps < 3) : (steps += 1) o -= 1;
- return o;
+ while (s > 0 and text[s] & 0xC0 == 0x80 and steps < 3) : (steps += 1) s -= 1;
+ // Only a lead byte whose sequence reaches `o`, its bytes all
+ // continuations, makes `o` the inside of a rune; a stray continuation
+ // byte (invalid UTF-8) is a place of its own.
+ const n = std.unicode.utf8ByteSequenceLength(text[s]) catch return o;
+ if (s + n <= o or s + n > text.len) return o;
+ for (text[s + 1 .. s + n]) |c| if (c & 0xC0 != 0x80) return o;
+ return s;
}
/// `off`, or the end of the rune it falls inside.
pub fn runeEnd(text: []const u8, off: usize) usize {
- var o = @min(off, text.len);
- var steps: u8 = 0;
- while (o < text.len and text[o] & 0xC0 == 0x80 and steps < 3) : (steps += 1) o += 1;
- return o;
+ const o = @min(off, text.len);
+ const s = runeStart(text, o);
+ if (s == o) return o;
+ return s + (std.unicode.utf8ByteSequenceLength(text[s]) catch 1);
}
/// The rune boundary after the one at `off`.
diff --git a/src/sam_edit.zig b/src/sam_edit.zig
index 29dec726..95d59a9d 100644
--- a/src/sam_edit.zig
+++ b/src/sam_edit.zig
@@ -16,6 +16,13 @@
const std = @import("std");
const regexp = @import("regexp.zig");
const addr_lang = @import("ninep/addr.zig");
+const modal = @import("modal.zig");
+
+/// The next place after `p` a search may start: a rune on, and past the
+/// text's end once there, which ends a loop over `p <= end`.
+fn stepRune(text: []const u8, p: usize) usize {
+ return if (p >= text.len) p + 1 else modal.nextRune(text, p);
+}
pub const Range = addr_lang.Range;
@@ -418,9 +425,16 @@ const Exec = struct {
};
}
+ /// A match covers the runes it touches, as an address's does
+ /// (addr.zig): mvzr matches bytes, so `.` would take one byte of `é`
+ /// and x, y and s would split it. From `from` on only, so a match never
+ /// reaches back into what the last one took.
fn find(ex: *Exec, rx: *regexp.Regex, from: usize, hi: usize) Failure!?regexp.Regex.Match {
if (from > hi) return null;
- return rx.find(ex.text, from, hi, hi) catch fail(ex.why, "{s}", .{addr_lang.e_slow});
+ const m = (rx.find(ex.text, from, hi, hi) catch return fail(ex.why, "{s}", .{addr_lang.e_slow})) orelse return null;
+ var start = modal.runeStart(ex.text, m.start);
+ if (start < from) start = modal.runeEnd(ex.text, m.start);
+ return .{ .start = start, .end = @max(start, modal.runeEnd(ex.text, m.end)) };
}
/// ecmd.c:62, cmdexec: runs `c` from `dot`, answering the dot it leaves.
@@ -516,10 +530,10 @@ const Exec = struct {
const m = (try ex.find(&rx, p1, r.q1)) orelse break;
if (m.start == m.end) {
if (op != null and m.start == op.?) {
- p1 += 1;
+ p1 = stepRune(ex.text, p1);
continue;
}
- p1 = m.end + 1;
+ p1 = stepRune(ex.text, m.end);
} else p1 = m.end;
op = m.end;
n -= 1;
@@ -560,10 +574,10 @@ const Exec = struct {
if (try ex.find(&rx, p, r.q1)) |m| {
if (m.start == m.end) {
if (op != null and m.start == op.?) {
- p += 1;
+ p = stepRune(ex.text, p);
continue;
}
- p = m.end + 1;
+ p = stepRune(ex.text, m.end);
} else p = m.end;
tr = if (xy) .{ .q0 = clip(m.start), .q1 = clip(m.end) } else .{ .q0 = clip(op.?), .q1 = clip(m.start) };
sel_end = m.end;
@@ -723,3 +737,14 @@ test "the dot an Edit leaves selects what a change put where it stood" {
try std.testing.expectEqual(Range{ .q0 = 2, .q1 = 5 }, moveDot(.{ .q0 = 2, .q1 = 2 }, &ops));
try std.testing.expectEqual(Range{ .q0 = 6, .q1 = 7 }, moveDot(.{ .q0 = 5, .q1 = 6 }, &ops));
}
+
+test "x, y and s match whole runes: a byte pattern never splits one" {
+ // é is c3 a9: `.` takes it whole, as it takes a lone combining mark.
+ try expectEdit("\u{e9}", ",x/./a/|/", "\u{e9}|");
+ try expectEdit("e\u{301}t", ",s/./X/g", "XXX");
+ try expectEdit("a\u{e9}b", ",y/\u{e9}/c/-/", "-\u{e9}-");
+ try expectEdit("\u{e9}\u{e9}", ",x/\\xC3/c/E/", "EE");
+ // An invalid byte is a place of its own.
+ try expectEdit("a\x81b", ",x/./a/|/", "a|\x81|b|");
+ try expectEdit("\u{e9}", ",s/$/!/", "\u{e9}!");
+}