summaryrefslogtreecommitdiff
path: root/src/modal.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/modal.zig')
-rw-r--r--src/modal.zig121
1 files changed, 119 insertions, 2 deletions
diff --git a/src/modal.zig b/src/modal.zig
index 953aa900..11eae743 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -886,7 +886,15 @@ pub fn nextGrapheme(text: []const u8, off: usize) usize {
if (off >= text.len) return text.len;
// The editor's own offsets are already boundaries. Keep the overwhelmingly
// common ASCII path O(1); only repair a continuation-byte input here.
- if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1;
+ //
+ // GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a
+ // following LF into the same cluster. `graphemeStart` spells that exclusion
+ // out (:875) and this did not, so the two disagreed about a CRLF file by
+ // exactly one byte — a head stepped onto the offset between CR and LF and
+ // `graphemeStart` then repaired it back onto the CR. Excluded here for the
+ // same reason and in the same words; everything else ASCII is still O(1).
+ if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80) and
+ !(text[off] == '\r' and off + 1 < text.len and text[off + 1] == '\n')) return off + 1;
var start = off;
while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1;
if (start != off) start = graphemeStart(text, off);
@@ -903,7 +911,10 @@ pub fn prevGrapheme(text: []const u8, off: usize) usize {
bounded = repaired;
}
if (bounded == 0) return 0;
- if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1;
+ // ...and the same GB3 exclusion, from the other side: a CR before this LF
+ // means the cluster starts one byte earlier than the fast path would say.
+ if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80) and
+ !(bounded >= 2 and text[bounded - 2] == '\r' and text[bounded - 1] == '\n')) return bounded - 1;
// Graphemes cannot cross a line break. Restrict the forward segmentation
// needed for a reverse step to the current line instead of rescanning the
// complete buffer.
@@ -1470,6 +1481,112 @@ test "extended grapheme boundaries cover combining emoji flag and CJK text" {
try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3));
}
+test "the ASCII arms of graphemeStart and nextGrapheme agree with the UAX #29 walk" {
+ // Both functions answer ASCII from arithmetic and hand everything else to the segmenter. The
+ // guard is a claim about UAX #29 (an ASCII scalar is its own cluster unless the next scalar
+ // extends it, and every extender is non-ASCII), so pin it against the walk it skips rather
+ // than against transcribed offsets: same text, both routes, every offset including past the end.
+ const H = struct {
+ // `graphemeStart` with the ASCII arm deleted — nothing else changed.
+ fn start(text: []const u8, off: usize) usize {
+ const bounded = @min(off, text.len);
+ if (bounded == text.len) return text.len;
+ var it = uucode.grapheme.utf8Iterator(text);
+ while (it.nextGrapheme()) |g| {
+ if (bounded < g.end) return g.start;
+ }
+ return text.len;
+ }
+ // `nextGrapheme` with the ASCII arm deleted.
+ fn next(text: []const u8, off: usize) usize {
+ if (off >= text.len) return text.len;
+ var s = off;
+ while (s > 0 and (text[s] & 0xC0) == 0x80) s -= 1;
+ if (s != off) s = start(text, off);
+ var it = uucode.grapheme.utf8Iterator(text[s..]);
+ const g = it.nextGrapheme() orelse return @min(s + 1, text.len);
+ return s + g.end;
+ }
+ fn check(text: []const u8) !void {
+ var off: usize = 0;
+ while (off <= text.len + 2) : (off += 1) {
+ std.testing.expectEqual(start(text, off), graphemeStart(text, off)) catch |e| {
+ std.debug.print("graphemeStart({any}, {d})\n", .{ text, off });
+ return e;
+ };
+ std.testing.expectEqual(next(text, off), nextGrapheme(text, off)) catch |e| {
+ std.debug.print("nextGrapheme({any}, {d})\n", .{ text, off });
+ return e;
+ };
+ }
+ }
+ };
+
+ // Scalars that extend a preceding ASCII base into ONE cluster, which is the whole reason the
+ // fast path inspects its neighbour: a combining mark, a ZWJ sequence, a spacing mark
+ // (Devanagari visarga), a variation selector. Plus wide glyphs, a regional-indicator pair,
+ // and three shapes of invalid UTF-8 the segmenter must still be trusted with: a bad start
+ // byte, a truncated tail, a bad continuation.
+ const neighbours = [_][]const u8{
+ "", "a", "\u{301}", "\u{200d}\u{1f680}",
+ "\u{903}", "\u{fe0f}", "\u{20e3}", "\u{4e16}\u{754c}",
+ "\u{1f642}", "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8",
+ "\xe4\x28\xb8",
+ };
+ // Every byte the range test can see, ASCII and not: 0x20..0x7e take the fast path, and \t, \r,
+ // the rest of the C0 controls and DEL are excluded by it and must still reach the same answer.
+ var buf: [16]u8 = undefined;
+ var b: u8 = 0;
+ while (b < 0x80) : (b += 1) {
+ buf[0] = b;
+ for (neighbours) |tail| {
+ @memcpy(buf[1..][0..tail.len], tail);
+ try H.check(buf[0 .. 1 + tail.len]);
+ // ...and the same byte as a follower, so a boundary is probed from both sides.
+ @memcpy(buf[0..tail.len], tail);
+ buf[tail.len] = b;
+ try H.check(buf[0 .. tail.len + 1]);
+ }
+ }
+
+ // Text that has no CR-LF pair in it: GB3 is the one ASCII-only rule that joins two clusters,
+ // and it gets its own test below because it is the single exclusion every fast path has to
+ // carry by hand.
+ for ([_][]const u8{ "a\r", "\ra", "\n\r", "a\rb\nc" }) |text| try H.check(text);
+
+ // Mixed text long enough that a fast-path run starts, ends and restarts inside one string.
+ try H.check("plain ascii then \u{4e16}\u{754c} then e\u{301} then more ascii");
+}
+
+// GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a following LF into the same
+// cluster. Each of the three steppers carries that exclusion separately - `graphemeStart` at :875,
+// `nextGrapheme`'s ASCII arm at :896, `prevGrapheme`'s at :916 - so nothing but a test keeps them
+// agreeing. The invariant is that all three answer the same CRLF boundary: for every cluster the
+// segmenter reports, `graphemeStart` maps its start to itself, `nextGrapheme` maps that start to
+// its end, and `prevGrapheme` maps its end back to the start.
+//
+// This was a live bug: `nextGrapheme` and `prevGrapheme` stepped exactly one byte whenever the
+// byte at the offset and its neighbour were ASCII, so on a CRLF file the flat-buffer range engine
+// could step a head to offset 1 and `graphemeStart` would repair that same offset back to 0. Both
+// arms now spell the exclusion out, and this test is what holds them there.
+test "GB3 keeps CR-LF one cluster for every grapheme step" {
+ const text = "a\r\nb";
+ // The reference: the same segmentation the slow arms of these functions run.
+ var it = uucode.grapheme.utf8Iterator(text);
+ var starts: [8]usize = undefined;
+ var ends: [8]usize = undefined;
+ var n: usize = 0;
+ while (it.nextGrapheme()) |g| : (n += 1) {
+ starts[n] = g.start;
+ ends[n] = g.end;
+ }
+ for (starts[0..n], ends[0..n]) |start, end| {
+ try std.testing.expectEqual(start, graphemeStart(text, start));
+ try std.testing.expectEqual(end, nextGrapheme(text, start));
+ try std.testing.expectEqual(start, prevGrapheme(text, end));
+ }
+}
+
test "Unicode find and word motion stay on grapheme boundaries" {
const lines = [_][]const u8{"\u{e9}x\u{e9}"};
try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'é', true, false, 1).?);