summaryrefslogtreecommitdiff
path: root/src/modal.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-18 12:09:12 -0300
committerGabriel Schneider <[email protected]>2026-08-18 23:45:25 -0300
commitc24a9e40215bc30c68c9db7675f1c6a07d9bbae3 (patch)
tree9afdb1f3d128bda3f3cf4837c2893982b0a8f69c /src/modal.zig
parent31fece62f56aa2311e2325de83659edbc9e641db (diff)
downloadpardes-c24a9e40215bc30c68c9db7675f1c6a07d9bbae3.tar.gz
pardes-c24a9e40215bc30c68c9db7675f1c6a07d9bbae3.zip
modal + file_pane: rework editing math and pane behavior, config/syntax additions, unit tests
Diffstat (limited to 'src/modal.zig')
-rw-r--r--src/modal.zig418
1 files changed, 286 insertions, 132 deletions
diff --git a/src/modal.zig b/src/modal.zig
index 6ab7ee35..a98cfeb4 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -1,12 +1,14 @@
const std = @import("std");
+const uucode = @import("uucode");
// Modal-editing text math, kept free of vaxis/ghostty so it can be unit-tested
// in isolation (see the `unit-test` build step). main.zig wires this onto the
// pane's cursor + (for file panes) its content.
//
-// The cursor sits ON a character: col is a char index in [0, line.len]; col ==
-// line.len means "on the line terminator / after the last char". Motions are
-// written to land on real characters; main.zig clamps for display.
+// The cursor sits ON a grapheme: col is its UTF-8 byte offset in
+// [0, line.len]; col == line.len means "on the line terminator / after the
+// last grapheme". Motions never leave a cursor in the middle of UTF-8 or an
+// extended grapheme cluster.
pub const Cursor = struct {
row: usize = 0,
@@ -21,23 +23,57 @@ pub const Cursor = struct {
// non-ws, ws = space/tab/newline).
pub const Kind = enum { word, punct, ws };
+fn codepointAt(text: []const u8, off: usize) u21 {
+ if (off >= text.len) return 0xFFFD;
+ const n = std.unicode.utf8ByteSequenceLength(text[off]) catch return 0xFFFD;
+ if (off + n > text.len) return 0xFFFD;
+ return std.unicode.utf8Decode(text[off .. off + n]) catch 0xFFFD;
+}
+
+fn isUnicodeWhitespace(cp: u21) bool {
+ if (cp == ' ' or (cp >= '\t' and cp <= '\r') or cp == 0x85) return true;
+ return switch (uucode.get(.general_category, cp)) {
+ .separator_space, .separator_line, .separator_paragraph => true,
+ else => false,
+ };
+}
+
+fn kindOfCodepoint(cp: u21) Kind {
+ if (isUnicodeWhitespace(cp)) return .ws;
+ if (cp == '_') return .word;
+ return switch (uucode.get(.general_category, cp)) {
+ .letter_uppercase,
+ .letter_lowercase,
+ .letter_titlecase,
+ .letter_modifier,
+ .letter_other,
+ .mark_nonspacing,
+ .mark_spacing_combining,
+ .mark_enclosing,
+ .number_decimal_digit,
+ .number_letter,
+ .number_other,
+ .punctuation_connector,
+ => .word,
+ else => .punct,
+ };
+}
+
pub fn kindOf(c: u8) Kind {
- if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws;
- if (std.ascii.isAlphanumeric(c) or c == '_') return .word;
- return .punct;
+ return kindOfCodepoint(c);
}
// "long word" (W/B/E): only whitespace separates; punct is part of a word.
-fn kindOfLong(c: u8) Kind {
- if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws;
- return .word;
+fn kindOfLong(cp: u21) Kind {
+ return if (isUnicodeWhitespace(cp)) .ws else .word;
}
fn kindAt(lines: []const []const u8, c: Cursor, long: bool) Kind {
if (c.row >= lines.len) return .ws;
const line = lines[c.row];
if (c.col >= line.len) return .ws; // line terminator / EOF = whitespace
- return if (long) kindOfLong(line[c.col]) else kindOf(line[c.col]);
+ const cp = codepointAt(line, graphemeStart(line, c.col));
+ return if (long) kindOfLong(cp) else kindOfCodepoint(cp);
}
fn lineLenOf(lines: []const []const u8, row: usize) usize {
@@ -51,7 +87,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool {
if (c.row >= lines.len) return false;
const llen = lineLenOf(lines, c.row);
if (c.col < llen) {
- c.col += 1;
+ c.col = nextGrapheme(lines[c.row], c.col);
return true;
}
// at the newline: move to next line start
@@ -65,7 +101,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool {
fn stepBwd(lines: []const []const u8, c: *Cursor) bool {
if (c.col > 0) {
- c.col -= 1;
+ c.col = prevGrapheme(lines[c.row], c.col);
return true;
}
if (c.row == 0) return false;
@@ -83,7 +119,7 @@ fn atEof(lines: []const []const u8, c: Cursor) bool {
pub fn firstNonWs(line: []const u8) usize {
var i: usize = 0;
- while (i < line.len and (line[i] == ' ' or line[i] == '\t')) i += 1;
+ while (i < line.len and isUnicodeWhitespace(codepointAt(line, i))) i = nextGrapheme(line, i);
return i;
}
@@ -95,7 +131,7 @@ pub fn lineStart(c: Cursor) Cursor {
pub fn lineEnd(lines: []const []const u8, c: Cursor) Cursor {
const llen = lineLenOf(lines, c.row);
- return .{ .row = c.row, .col = if (llen == 0) 0 else llen - 1 };
+ return .{ .row = c.row, .col = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen) };
}
pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor {
@@ -107,28 +143,33 @@ pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor {
// ---- char/line motions ----
-pub fn charLeft(c: Cursor) Cursor {
- return .{ .row = c.row, .col = if (c.col > 0) c.col - 1 else 0 };
+pub fn charLeft(lines: []const []const u8, c: Cursor) Cursor {
+ if (c.row >= lines.len) return .{ .row = c.row, .col = 0 };
+ return .{ .row = c.row, .col = prevGrapheme(lines[c.row], @min(c.col, lines[c.row].len)) };
}
pub fn charRight(lines: []const []const u8, c: Cursor) Cursor {
const llen = lineLenOf(lines, c.row);
- const last = if (llen == 0) 0 else llen - 1;
- return .{ .row = c.row, .col = if (c.col < last) c.col + 1 else last };
+ const last = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen);
+ return .{ .row = c.row, .col = if (c.col < last) @min(nextGrapheme(lines[c.row], c.col), last) else last };
+}
+
+fn clampLineCol(line: []const u8, col: usize) usize {
+ if (line.len == 0) return 0;
+ const last = prevGrapheme(line, line.len);
+ return graphemeStart(line, @min(col, last));
}
pub fn lineDown(lines: []const []const u8, c: Cursor) Cursor {
const nr = if (c.row + 1 < lines.len) c.row + 1 else c.row;
const llen = lineLenOf(lines, nr);
- const last = if (llen == 0) 0 else llen - 1;
- return .{ .row = nr, .col = if (c.col < last) c.col else last };
+ return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) };
}
pub fn lineUp(lines: []const []const u8, c: Cursor) Cursor {
const nr = if (c.row > 0) c.row - 1 else c.row;
const llen = lineLenOf(lines, nr);
- const last = if (llen == 0) 0 else llen - 1;
- return .{ .row = nr, .col = if (c.col < last) c.col else last };
+ return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) };
}
// ---- word motions ----
@@ -205,19 +246,22 @@ pub fn gotoLast(lines: []const []const u8) Cursor {
}
// the character the cursor sits on; line terminators / EOF read as '\n'.
-fn charAt(lines: []const []const u8, c: Cursor) u8 {
+fn codepointAtCursor(lines: []const []const u8, c: Cursor) u21 {
if (c.row >= lines.len) return '\n';
const line = lines[c.row];
if (c.col >= line.len) return '\n';
- return line[c.col];
+ return codepointAt(line, graphemeStart(line, c.col));
+}
+
+fn charAt(lines: []const []const u8, c: Cursor) u8 {
+ const cp = codepointAtCursor(lines, c);
+ return if (cp <= 0x7f) @intCast(cp) else 0;
}
// `f`/`F`/`t`/`T`: the nth occurrence of `ch` after/before the cursor, across
// line boundaries (helix: not confined to the line). `till` stops one position
// short of the hit. Returns null (no move) when there aren't n occurrences.
pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: bool, n: usize) ?Cursor {
- if (ch > 0x7f) return null; // ponytail: ASCII targets only (byte columns)
- const target: u8 = @intCast(ch);
var p = clampToChar(lines, c);
var left = if (n == 0) 1 else n;
while (left > 0) {
@@ -226,7 +270,7 @@ pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till:
} else {
if (!stepBwd(lines, &p)) return null;
}
- if (charAt(lines, p) == target) left -= 1;
+ if (codepointAtCursor(lines, p) == ch) left -= 1;
}
if (till) {
if (fwd) _ = stepBwd(lines, &p) else _ = stepFwd(lines, &p);
@@ -369,16 +413,32 @@ pub fn wordRange(lines: []const []const u8, c0: Cursor, long: bool, around: bool
const k = kindAt(lines, c, long);
if (k == .ws) return null;
var lo = c.col;
- while (lo > 0 and kindAt(lines, .{ .row = c.row, .col = lo - 1 }, long) == k) lo -= 1;
+ while (lo > 0) {
+ const prev = prevGrapheme(line, lo);
+ if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != k) break;
+ lo = prev;
+ }
var hi = c.col;
- while (hi + 1 < line.len and kindAt(lines, .{ .row = c.row, .col = hi + 1 }, long) == k) hi += 1;
+ while (true) {
+ const next = nextGrapheme(line, hi);
+ if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != k) break;
+ hi = next;
+ }
if (around) {
var h2 = hi;
- while (h2 + 1 < line.len and (line[h2 + 1] == ' ' or line[h2 + 1] == '\t')) h2 += 1;
+ while (true) {
+ const next = nextGrapheme(line, h2);
+ if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != .ws) break;
+ h2 = next;
+ }
if (h2 != hi) {
hi = h2;
} else {
- while (lo > 0 and (line[lo - 1] == ' ' or line[lo - 1] == '\t')) lo -= 1;
+ while (lo > 0) {
+ const prev = prevGrapheme(line, lo);
+ if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != .ws) break;
+ lo = prev;
+ }
}
}
return .{ .a = .{ .row = c.row, .col = lo }, .b = .{ .row = c.row, .col = hi } };
@@ -403,7 +463,7 @@ pub fn paragraphRange(lines: []const []const u8, c0: Cursor, around: bool) ?Rang
}
}
const llen = lineLenOf(lines, r1);
- return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else llen - 1 } };
+ return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else prevGrapheme(lines[r1], llen) } };
}
// ---- helpers used by motions + main.zig ----
@@ -415,7 +475,7 @@ pub fn clampToChar(lines: []const []const u8, c: Cursor) Cursor {
}
const llen = lineLenOf(lines, c.row);
if (llen == 0) return .{ .row = c.row, .col = 0 };
- return .{ .row = c.row, .col = @min(c.col, llen - 1) };
+ return .{ .row = c.row, .col = clampLineCol(lines[c.row], c.col) };
}
pub fn lineCount(content: []const u8) usize {
@@ -465,8 +525,9 @@ pub fn insertAt(alloc: std.mem.Allocator, content: []const u8, c: Cursor, text:
pub fn deleteChar(alloc: std.mem.Allocator, content: []const u8, c: Cursor) ![]u8 {
const line = lineSlice(content, c.row);
if (c.col >= line.len) return alloc.dupe(u8, content);
- const off = lineStartOffset(content, c.row) + c.col;
- return spliceAlloc(alloc, content, off, off + 1, "");
+ const col = graphemeStart(line, c.col);
+ const line_start = lineStartOffset(content, c.row);
+ return spliceAlloc(alloc, content, line_start + col, line_start + nextGrapheme(line, col), "");
}
// delete whole lines [r0, r1] inclusive (the line content + their terminators).
@@ -520,9 +581,13 @@ fn rangeBytes(content: []const u8, a: Cursor, b: Cursor) struct { s: usize, e: u
lo = b;
hi = a;
}
- const s = lineStartOffset(content, lo.row) + @min(lo.col, lineSlice(content, lo.row).len);
- var e = lineStartOffset(content, hi.row) + @min(hi.col, lineSlice(content, hi.row).len);
- if (e < content.len) e += 1; // include the char under the head
+ const lo_line = lineSlice(content, lo.row);
+ const hi_line = lineSlice(content, hi.row);
+ const lo_col = graphemeStart(lo_line, @min(lo.col, lo_line.len));
+ const hi_col = graphemeStart(hi_line, @min(hi.col, hi_line.len));
+ const s = lineStartOffset(content, lo.row) + lo_col;
+ var e = lineStartOffset(content, hi.row) + hi_col;
+ if (e < content.len) e = nextGrapheme(content, e); // include the grapheme under the head
return .{ .s = s, .e = @max(s, e) };
}
@@ -593,10 +658,22 @@ pub fn replaceRange(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b:
// `r<ch>`: overwrite every char in the inclusive range [a, b] with `ch` β€”
// NEWLINES TOO (helix replace maps every grapheme, so `xrz` joins lines).
-pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u8) ![]u8 {
+pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u21) ![]u8 {
const r = rangeBytes(content, a, b);
- const out = try alloc.dupe(u8, content);
- for (out[r.s..r.e]) |*p| p.* = ch;
+ var encoded: [4]u8 = undefined;
+ const encoded_len = try std.unicode.utf8Encode(ch, &encoded);
+ var count: usize = 0;
+ var at = r.s;
+ while (at < r.e) : (count += 1) at = nextGrapheme(content, at);
+ const replacement_len = count * encoded_len;
+ const out = try alloc.alloc(u8, content.len - (r.e - r.s) + replacement_len);
+ @memcpy(out[0..r.s], content[0..r.s]);
+ var write = r.s;
+ for (0..count) |_| {
+ @memcpy(out[write..][0..encoded_len], encoded[0..encoded_len]);
+ write += encoded_len;
+ }
+ @memcpy(out[write..], content[r.e..]);
return out;
}
@@ -729,9 +806,9 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C
//
// Gap-offset ranges over the FLAT buffer, ported faithfully from
// helix-core/src/movement.rs + selection.rs @ 278b24389 (the genizah
-// checkout). Positions are gap offsets 0..=text.len β€” "char indices" in
-// helix terms, bytes here (ASCII-exact; UTF-8 stepped by sequence, matching
-// the byte-column convention of the rest of pardes). A range with
+// checkout). Positions are UTF-8 gap offsets 0..=text.len. Stored columns
+// remain byte offsets, but every range boundary is an extended-grapheme
+// boundary. A range with
// head > anchor selects [anchor, head) with the block cursor ON head-1;
// head < anchor selects [head, anchor) with the cursor ON head. The
// differential suite (test/hxcases, `zig build hxdiff`) pins every behavior
@@ -739,19 +816,58 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C
pub const HxRange = struct { anchor: usize, head: usize };
-/// one grapheme forward (UTF-8 sequence step), clamped at text.len
+/// The grapheme containing `off`, or text.len at EOF. This is also the repair
+/// path for stale/external byte columns that happen to point into UTF-8.
+pub fn graphemeStart(text: []const u8, off: usize) usize {
+ const bounded = @min(off, text.len);
+ if (bounded == text.len) return text.len;
+ var it = uucode.grapheme.utf8Iterator(text);
+ while (it.nextGrapheme()) |g| {
+ if (bounded < g.end) return g.start;
+ }
+ return text.len;
+}
+
+/// one extended grapheme forward, clamped at text.len
pub fn nextGrapheme(text: []const u8, off: usize) usize {
if (off >= text.len) return text.len;
- var o = off + 1;
- while (o < text.len and (text[o] & 0xC0) == 0x80) o += 1;
- return o;
+ // The editor's own offsets are already boundaries. Keep the overwhelmingly
+ // common ASCII path O(1); only repair a continuation-byte input here.
+ if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1;
+ var start = off;
+ while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1;
+ if (start != off) start = graphemeStart(text, off);
+ var it = uucode.grapheme.utf8Iterator(text[start..]);
+ const g = it.nextGrapheme() orelse return @min(start + 1, text.len);
+ return start + g.end;
}
pub fn prevGrapheme(text: []const u8, off: usize) usize {
- if (off == 0) return 0;
- var o = off - 1;
- while (o > 0 and (text[o] & 0xC0) == 0x80) o -= 1;
- return o;
+ var bounded = @min(off, text.len);
+ if (bounded < text.len) {
+ const repaired = graphemeStart(text, bounded);
+ if (repaired < bounded) return repaired;
+ bounded = repaired;
+ }
+ if (bounded == 0) return 0;
+ if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1;
+ // Graphemes cannot cross a line break. Restrict the forward segmentation
+ // needed for a reverse step to the current line instead of rescanning the
+ // complete buffer.
+ const line_start = if (std.mem.lastIndexOfScalar(u8, text[0 .. bounded - 1], '\n')) |nl| nl + 1 else 0;
+ if (line_start == bounded) return bounded - 1; // the newline is its own editor cell
+ var it = uucode.grapheme.utf8Iterator(text[line_start..bounded]);
+ var last = line_start;
+ while (it.nextGrapheme()) |g| last = line_start + g.start;
+ return last;
+}
+
+/// Byte offset of the zero-based grapheme column, clamped to the line end.
+pub fn graphemeAtColumn(text: []const u8, column: usize) usize {
+ var at: usize = 0;
+ var col: usize = 0;
+ while (at < text.len and col < column) : (col += 1) at = nextGrapheme(text, at);
+ return at;
}
/// the block cursor cell of a range (helix Range::cursor)
@@ -792,14 +908,16 @@ pub fn hxLineEndIdx(text: []const u8, line: usize) usize {
/// gap offset -> (row, col) cell
pub fn hxPos(text: []const u8, off: usize) Cursor {
- const o = @min(off, text.len);
+ const bounded = @min(off, text.len);
// The line start is the byte after the last '\n' BEFORE off, which is the
// same number lineStartOffset(text, row) walks the whole prefix to reach β€”
// one backward scan of a single line instead of a second pass over
// everything above the cursor. On a multi-MB buffer that second pass was
// most of what a keystroke cost.
- const s = if (std.mem.lastIndexOfScalar(u8, text[0..o], '\n')) |nl| nl + 1 else 0;
- return .{ .row = hxLineOf(text, off), .col = o - s };
+ const s = if (std.mem.lastIndexOfScalar(u8, text[0..bounded], '\n')) |nl| nl + 1 else 0;
+ const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len;
+ const col = graphemeStart(text[s..e], @min(bounded - s, e - s));
+ return .{ .row = hxLineOf(text, s + col), .col = col };
}
/// (row, col) -> clamped gap offset; col == line length lands ON the '\n'
@@ -809,7 +927,8 @@ pub fn hxOff(text: []const u8, c: Cursor) usize {
// hxLineEndIdx(text, row) inlined: it starts by walking to `row` again,
// and we are already standing there
const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len;
- return @min(s + c.col, e);
+ const raw = @min(c.col, e - s);
+ return s + graphemeStart(text[s..e], raw);
}
pub const WordTarget = enum {
@@ -827,35 +946,36 @@ pub const WordTarget = enum {
// that distinction is load-bearing in reached_target.
const HxCat = enum { word, punct, ws, eol };
-fn hxCat(b: u8) HxCat {
- if (b == '\n' or b == '\r') return .eol;
- if (b == ' ' or b == '\t' or b == 0x0b or b == 0x0c) return .ws;
- if (std.ascii.isAlphanumeric(b) or b == '_' or b >= 0x80) return .word;
- return .punct;
+fn hxCatAt(text: []const u8, off: usize) HxCat {
+ if (off >= text.len) return .eol;
+ const cp = codepointAt(text, off);
+ if (cp == '\n' or cp == '\r') return .eol;
+ return switch (kindOfCodepoint(cp)) {
+ .word => .word,
+ .punct => .punct,
+ .ws => .ws,
+ };
}
-fn hxIsWs(b: u8) bool { // Rust char::is_whitespace (includes line endings)
- const c = hxCat(b);
+fn hxIsWs(c: HxCat) bool { // Rust char::is_whitespace (includes line endings)
return c == .ws or c == .eol;
}
-fn hxIsWordBoundary(a: u8, b: u8) bool {
- return hxCat(a) != hxCat(b);
+fn hxIsWordBoundary(a: HxCat, b: HxCat) bool {
+ return a != b;
}
-fn hxIsLongBoundary(a: u8, b: u8) bool {
- const ca = hxCat(a);
- const cb = hxCat(b);
- if ((ca == .word and cb == .punct) or (ca == .punct and cb == .word)) return false;
- return ca != cb;
+fn hxIsLongBoundary(a: HxCat, b: HxCat) bool {
+ if ((a == .word and b == .punct) or (a == .punct and b == .word)) return false;
+ return a != b;
}
-fn hxReached(target: WordTarget, prev: u8, next: u8) bool {
+fn hxReached(target: WordTarget, prev: HxCat, next: HxCat) bool {
return switch (target) {
- .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)),
- .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol),
- .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)),
- .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol),
+ .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (next == .eol or !hxIsWs(next)),
+ .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or next == .eol),
+ .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (next == .eol or !hxIsWs(next)),
+ .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or next == .eol),
};
}
@@ -896,45 +1016,35 @@ fn hxRangeToTarget(text: []const u8, target: WordTarget, origin: HxRange, is_pre
var anchor = origin.anchor;
var head = origin.head;
var it = origin.head;
- var prev_ch: ?u8 = if (is_prev)
- (if (it < text.len) text[it] else null)
+ var prev_cat: ?HxCat = if (is_prev)
+ (if (it < text.len) hxCatAt(text, it) else null)
else
- (if (it > 0) text[it - 1] else null);
+ (if (it > 0) hxCatAt(text, prevGrapheme(text, it)) else null);
// skip any initial newline characters
while (true) {
- const ch: u8 = if (is_prev) blk: {
- if (it == 0) break;
- break :blk text[it - 1];
- } else blk: {
- if (it >= text.len) break;
- break :blk text[it];
- };
- if (ch != '\n' and ch != '\r') break;
- if (is_prev) it -= 1 else it += 1;
- prev_ch = ch;
- if (is_prev) head -|= 1 else head += 1;
+ if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break;
+ const cell = if (is_prev) prevGrapheme(text, it) else it;
+ const cat = hxCatAt(text, cell);
+ if (cat != .eol) break;
+ it = if (is_prev) cell else nextGrapheme(text, cell);
+ prev_cat = cat;
+ head = it;
}
- if (prev_ch != null and hxCat(prev_ch.?) == .eol) anchor = head;
+ if (prev_cat == .eol) anchor = head;
// find the target position
const head_start = head;
while (true) {
- const next_ch: u8 = if (is_prev) blk: {
- if (it == 0) break;
- it -= 1;
- break :blk text[it];
- } else blk: {
- if (it >= text.len) break;
- const c = text[it];
- it += 1;
- break :blk c;
- };
- if (prev_ch == null or hxReached(target, prev_ch.?, next_ch)) {
+ if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break;
+ const cell = if (is_prev) prevGrapheme(text, it) else it;
+ const next_cat = hxCatAt(text, cell);
+ if (prev_cat == null or hxReached(target, prev_cat.?, next_cat)) {
if (head == head_start) anchor = head else break;
}
- prev_ch = next_ch;
- if (is_prev) head -|= 1 else head += 1;
+ prev_cat = next_cat;
+ it = if (is_prev) cell else nextGrapheme(text, cell);
+ head = it;
}
return .{ .anchor = anchor, .head = head };
}
@@ -1007,31 +1117,31 @@ pub fn hxVertTarget(text: []const u8, pos: usize, down: bool, count: usize, goal
const s = lineStartOffset(text, nline);
// hxLineEndIdx(text, nline) without its second walk to nline (see hxOff)
const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len;
- return @min(s + goal_col, e);
+ return s + graphemeStart(text[s..e], @min(goal_col, e - s));
}
/// f/F/t/T target cell. helix find_char: the exclusive (till) search starts
/// one further out so repeats make progress; not-found = null (no move).
-pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bool, count: usize) ?usize {
+pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u21, fwd: bool, till: bool, count: usize) ?usize {
var left = @max(1, count);
if (fwd) {
const head = nextGrapheme(text, cursor);
- var i = if (till) head + 1 else head;
+ var i = if (till) nextGrapheme(text, head) else head;
if (i > text.len) return null;
- while (i < text.len) : (i += 1) {
- if (text[i] == ch) {
+ while (i < text.len) : (i = nextGrapheme(text, i)) {
+ if (codepointAt(text, i) == ch) {
left -= 1;
- if (left == 0) return if (till) i - 1 else i;
+ if (left == 0) return if (till) prevGrapheme(text, i) else i;
}
}
return null;
}
- var i = if (till) cursor -| 1 else cursor;
+ var i = if (till) prevGrapheme(text, cursor) else cursor;
while (i > 0) {
- i -= 1;
- if (text[i] == ch) {
+ i = prevGrapheme(text, i);
+ if (codepointAt(text, i) == ch) {
left -= 1;
- if (left == 0) return if (till) i + 1 else i;
+ if (left == 0) return if (till) nextGrapheme(text, i) else i;
}
}
return null;
@@ -1040,26 +1150,19 @@ pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bo
// helix textobject.rs find_word_boundary
fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usize {
var prev: HxCat = if (fwd)
- (if (pos0 == 0) .ws else hxCat(text[pos0 - 1]))
+ (if (pos0 == 0) .ws else hxCatAt(text, prevGrapheme(text, pos0)))
else
- (if (pos0 >= text.len) .ws else hxCat(text[pos0]));
+ (if (pos0 >= text.len) .ws else hxCatAt(text, pos0));
var pos = pos0;
var it = pos0;
while (true) {
- const ch: u8 = if (fwd) blk: {
- if (it >= text.len) break;
- const c = text[it];
- it += 1;
- break :blk c;
- } else blk: {
- if (it == 0) break;
- it -= 1;
- break :blk text[it];
- };
- const cat = hxCat(ch);
+ if ((fwd and it >= text.len) or (!fwd and it == 0)) break;
+ const cell = if (fwd) it else prevGrapheme(text, it);
+ const cat = hxCatAt(text, cell);
if (cat == .eol or cat == .ws) return pos;
if (!long and cat != prev and pos != 0 and pos != text.len) return pos;
- if (fwd) pos += 1 else pos -|= 1;
+ it = if (fwd) nextGrapheme(text, cell) else cell;
+ pos = it;
prev = cat;
}
return pos;
@@ -1070,14 +1173,20 @@ fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usiz
pub fn hxTextobjectWord(text: []const u8, r: HxRange, around: bool, long: bool) HxRange {
const pos = hxCursor(text, r);
const word_start = hxFindWordBoundary(text, pos, false, long);
- const cat: HxCat = if (pos < text.len) hxCat(text[pos]) else .ws;
- const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, pos + 1, true, long);
+ const cat: HxCat = if (pos < text.len) hxCatAt(text, pos) else .ws;
+ const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, nextGrapheme(text, pos), true, long);
if (word_start == word_end or !around) return .{ .anchor = word_start, .head = word_end };
var end = word_end;
- while (end < text.len and hxIsWs(text[end]) and hxCat(text[end]) != .eol) end += 1;
+ while (end < text.len and hxIsWs(hxCatAt(text, end)) and hxCatAt(text, end) != .eol)
+ end = nextGrapheme(text, end);
if (end > word_end) return .{ .anchor = word_start, .head = end };
var start = word_start;
- while (start > 0 and hxIsWs(text[start - 1]) and hxCat(text[start - 1]) != .eol) start -= 1;
+ while (start > 0) {
+ const before = prevGrapheme(text, start);
+ const before_cat = hxCatAt(text, before);
+ if (!hxIsWs(before_cat) or before_cat == .eol) break;
+ start = before;
+ }
return .{ .anchor = start, .head = word_end };
}
@@ -1294,6 +1403,38 @@ test "hx find targets" {
try std.testing.expectEqual(@as(?usize, null), hxFindTarget(t, 0, 'z', true, false, 1));
}
+test "extended grapheme boundaries cover combining emoji flag and CJK text" {
+ const text = "a" ++ "e\u{301}" ++ "πŸ‘©πŸ½\u{200d}πŸš€" ++ "πŸ‡§πŸ‡·" ++ "η•Œ";
+ const boundaries = [_]usize{ 0, 1, 4, 19, 27, 30 };
+ for (boundaries[0 .. boundaries.len - 1], boundaries[1..]) |start, end| {
+ try std.testing.expectEqual(end, nextGrapheme(text, start));
+ try std.testing.expectEqual(start, prevGrapheme(text, end));
+ }
+ // Stale byte offsets are repaired to a cluster boundary instead of being
+ // allowed to leak continuation bytes into cursor state.
+ try std.testing.expectEqual(@as(usize, 1), graphemeStart(text, 2));
+ try std.testing.expectEqual(@as(usize, 1), prevGrapheme(text, 3));
+ try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3));
+}
+
+test "Unicode find and word motion stay on grapheme boundaries" {
+ const lines = [_][]const u8{"\u{e9}x\u{e9}"};
+ try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'Γ©', true, false, 1).?);
+
+ const text = "cafΓ© δΈ–η•Œ ok\n";
+ const first = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 1, .next_word_start);
+ try std.testing.expectEqual(@as(usize, 0), first.anchor);
+ try std.testing.expectEqual(@as(usize, 6), first.head);
+ const second = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 2, .next_word_start);
+ try std.testing.expectEqual(@as(usize, 6), second.anchor);
+ try std.testing.expectEqual(@as(usize, 13), second.head);
+
+ // Long-word motions split on Unicode whitespace, not only ASCII spaces.
+ const nbsp = "alpha\u{a0}beta\n";
+ const long = hxWordMove(nbsp, .{ .anchor = 0, .head = 1 }, 1, .next_long_word_start);
+ try std.testing.expectEqual(@as(usize, 7), long.head);
+}
+
test "hx increment" {
const a = std.testing.allocator;
{
@@ -1335,7 +1476,7 @@ test "kindOf" {
test "char/line motions" {
const lines = [_][]const u8{ "alpha beta", " two words", "x" };
const c = Cursor{ .row = 0, .col = 5 };
- try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(c));
+ try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(&.{"hello"}, c));
try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, charRight(&lines, c));
try std.testing.expectEqual(Cursor{ .row = 1, .col = 5 }, lineDown(&lines, c));
try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, lineUp(&lines, Cursor{ .row = 1, .col = 5 }));
@@ -1440,6 +1581,19 @@ test "deleteChar" {
try std.testing.expectEqualStrings("abc", r2);
}
+test "Unicode edits replace and delete whole graphemes" {
+ const a = std.testing.allocator;
+ const content = "A" ++ "e\u{301}" ++ "πŸ‘©πŸ½\u{200d}πŸš€" ++ "πŸ‡§πŸ‡·" ++ "η•Œ" ++ "Z";
+
+ const deleted = try deleteChar(a, content, .{ .row = 0, .col = 2 });
+ defer a.free(deleted);
+ try std.testing.expectEqualStrings("AπŸ‘©πŸ½\u{200d}πŸš€πŸ‡§πŸ‡·η•ŒZ", deleted);
+
+ const replaced = try replaceChars(a, content, .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 19 }, 'η•Œ');
+ defer a.free(replaced);
+ try std.testing.expectEqualStrings("Aη•Œη•Œη•Œη•ŒZ", replaced);
+}
+
test "deleteLines middle" {
const content = "one\ntwo\nthree\nfour";
const a = std.testing.allocator;