summaryrefslogtreecommitdiff
path: root/src/modal.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/modal.zig')
-rw-r--r--src/modal.zig26
1 files changed, 23 insertions, 3 deletions
diff --git a/src/modal.zig b/src/modal.zig
index f8093ed8..953aa900 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -499,7 +499,9 @@ pub fn lineStartOffset(content: []const u8, row: usize) usize {
pub fn lineSlice(content: []const u8, row: usize) []const u8 {
const start = lineStartOffset(content, row);
if (start >= content.len) return "";
- const nl = std.mem.indexOfPos(u8, content, start, "\n") orelse content.len;
+ // indexOfScalarPos, not indexOfPos with a one-byte needle: the latter runs the generic
+ // substring search where a memchr will do, and this is called once per visible row per frame.
+ const nl = std.mem.indexOfScalarPos(u8, content, start, '\n') orelse content.len;
return content[start..nl];
}
@@ -849,11 +851,29 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C
pub const HxRange = struct { anchor: usize, head: usize };
-/// The grapheme containing `off`, or text.len at EOF. This is also the repair
-/// path for stale/external byte columns that happen to point into UTF-8.
+/// The first byte of the grapheme cluster containing `off`.
+///
+/// The general answer needs UAX #29, which is why the slow path below iterates from the start of
+/// `text` with the full break state machine - and that made this the single hottest function in a
+/// keystroke: 21.5% of a profiled edit at the ESP32-P4's 40x12 geometry, because the render path
+/// calls it once per visible row with a column offset, so the cost follows the cursor's distance
+/// along its line. That is exactly the shape measured on the die, where inserting at column 320 of
+/// a fixed line cost 7.8 ms more than inserting at column 0 of the same line.
+///
+/// The fast path is sound rather than approximate. In UAX #29 every ASCII scalar is its own
+/// grapheme cluster with ONE exception, GB3: CR is joined to a following LF. Every other rule that
+/// could extend a cluster across `off` - Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator -
+/// is spelled with non-ASCII scalars. So if the byte at `off` and the byte before it are both
+/// ASCII and are not that CR-LF pair, `off` already IS a cluster boundary and there is nothing to
+/// search for. Text that is not all ASCII still takes the slow path, byte for byte as before.
pub fn graphemeStart(text: []const u8, off: usize) usize {
const bounded = @min(off, text.len);
if (bounded == text.len) return text.len;
+ if (text[bounded] < 0x80) {
+ if (bounded == 0) return 0;
+ const prev = text[bounded - 1];
+ if (prev < 0x80 and !(prev == '\r' and text[bounded] == '\n')) return bounded;
+ }
var it = uucode.grapheme.utf8Iterator(text);
while (it.nextGrapheme()) |g| {
if (bounded < g.end) return g.start;