diff options
Diffstat (limited to 'src/modal.zig')
| -rw-r--r-- | src/modal.zig | 418 |
1 files changed, 286 insertions, 132 deletions
diff --git a/src/modal.zig b/src/modal.zig index 6ab7ee35..a98cfeb4 100644 --- a/src/modal.zig +++ b/src/modal.zig @@ -1,12 +1,14 @@ const std = @import("std"); +const uucode = @import("uucode"); // Modal-editing text math, kept free of vaxis/ghostty so it can be unit-tested // in isolation (see the `unit-test` build step). main.zig wires this onto the // pane's cursor + (for file panes) its content. // -// The cursor sits ON a character: col is a char index in [0, line.len]; col == -// line.len means "on the line terminator / after the last char". Motions are -// written to land on real characters; main.zig clamps for display. +// The cursor sits ON a grapheme: col is its UTF-8 byte offset in +// [0, line.len]; col == line.len means "on the line terminator / after the +// last grapheme". Motions never leave a cursor in the middle of UTF-8 or an +// extended grapheme cluster. pub const Cursor = struct { row: usize = 0, @@ -21,23 +23,57 @@ pub const Cursor = struct { // non-ws, ws = space/tab/newline). pub const Kind = enum { word, punct, ws }; +fn codepointAt(text: []const u8, off: usize) u21 { + if (off >= text.len) return 0xFFFD; + const n = std.unicode.utf8ByteSequenceLength(text[off]) catch return 0xFFFD; + if (off + n > text.len) return 0xFFFD; + return std.unicode.utf8Decode(text[off .. off + n]) catch 0xFFFD; +} + +fn isUnicodeWhitespace(cp: u21) bool { + if (cp == ' ' or (cp >= '\t' and cp <= '\r') or cp == 0x85) return true; + return switch (uucode.get(.general_category, cp)) { + .separator_space, .separator_line, .separator_paragraph => true, + else => false, + }; +} + +fn kindOfCodepoint(cp: u21) Kind { + if (isUnicodeWhitespace(cp)) return .ws; + if (cp == '_') return .word; + return switch (uucode.get(.general_category, cp)) { + .letter_uppercase, + .letter_lowercase, + .letter_titlecase, + .letter_modifier, + .letter_other, + .mark_nonspacing, + .mark_spacing_combining, + .mark_enclosing, + .number_decimal_digit, + .number_letter, + .number_other, + .punctuation_connector, + => .word, + else => .punct, + }; +} + pub fn kindOf(c: u8) Kind { - if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws; - if (std.ascii.isAlphanumeric(c) or c == '_') return .word; - return .punct; + return kindOfCodepoint(c); } // "long word" (W/B/E): only whitespace separates; punct is part of a word. -fn kindOfLong(c: u8) Kind { - if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws; - return .word; +fn kindOfLong(cp: u21) Kind { + return if (isUnicodeWhitespace(cp)) .ws else .word; } fn kindAt(lines: []const []const u8, c: Cursor, long: bool) Kind { if (c.row >= lines.len) return .ws; const line = lines[c.row]; if (c.col >= line.len) return .ws; // line terminator / EOF = whitespace - return if (long) kindOfLong(line[c.col]) else kindOf(line[c.col]); + const cp = codepointAt(line, graphemeStart(line, c.col)); + return if (long) kindOfLong(cp) else kindOfCodepoint(cp); } fn lineLenOf(lines: []const []const u8, row: usize) usize { @@ -51,7 +87,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool { if (c.row >= lines.len) return false; const llen = lineLenOf(lines, c.row); if (c.col < llen) { - c.col += 1; + c.col = nextGrapheme(lines[c.row], c.col); return true; } // at the newline: move to next line start @@ -65,7 +101,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool { fn stepBwd(lines: []const []const u8, c: *Cursor) bool { if (c.col > 0) { - c.col -= 1; + c.col = prevGrapheme(lines[c.row], c.col); return true; } if (c.row == 0) return false; @@ -83,7 +119,7 @@ fn atEof(lines: []const []const u8, c: Cursor) bool { pub fn firstNonWs(line: []const u8) usize { var i: usize = 0; - while (i < line.len and (line[i] == ' ' or line[i] == '\t')) i += 1; + while (i < line.len and isUnicodeWhitespace(codepointAt(line, i))) i = nextGrapheme(line, i); return i; } @@ -95,7 +131,7 @@ pub fn lineStart(c: Cursor) Cursor { pub fn lineEnd(lines: []const []const u8, c: Cursor) Cursor { const llen = lineLenOf(lines, c.row); - return .{ .row = c.row, .col = if (llen == 0) 0 else llen - 1 }; + return .{ .row = c.row, .col = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen) }; } pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor { @@ -107,28 +143,33 @@ pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor { // ---- char/line motions ---- -pub fn charLeft(c: Cursor) Cursor { - return .{ .row = c.row, .col = if (c.col > 0) c.col - 1 else 0 }; +pub fn charLeft(lines: []const []const u8, c: Cursor) Cursor { + if (c.row >= lines.len) return .{ .row = c.row, .col = 0 }; + return .{ .row = c.row, .col = prevGrapheme(lines[c.row], @min(c.col, lines[c.row].len)) }; } pub fn charRight(lines: []const []const u8, c: Cursor) Cursor { const llen = lineLenOf(lines, c.row); - const last = if (llen == 0) 0 else llen - 1; - return .{ .row = c.row, .col = if (c.col < last) c.col + 1 else last }; + const last = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen); + return .{ .row = c.row, .col = if (c.col < last) @min(nextGrapheme(lines[c.row], c.col), last) else last }; +} + +fn clampLineCol(line: []const u8, col: usize) usize { + if (line.len == 0) return 0; + const last = prevGrapheme(line, line.len); + return graphemeStart(line, @min(col, last)); } pub fn lineDown(lines: []const []const u8, c: Cursor) Cursor { const nr = if (c.row + 1 < lines.len) c.row + 1 else c.row; const llen = lineLenOf(lines, nr); - const last = if (llen == 0) 0 else llen - 1; - return .{ .row = nr, .col = if (c.col < last) c.col else last }; + return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) }; } pub fn lineUp(lines: []const []const u8, c: Cursor) Cursor { const nr = if (c.row > 0) c.row - 1 else c.row; const llen = lineLenOf(lines, nr); - const last = if (llen == 0) 0 else llen - 1; - return .{ .row = nr, .col = if (c.col < last) c.col else last }; + return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) }; } // ---- word motions ---- @@ -205,19 +246,22 @@ pub fn gotoLast(lines: []const []const u8) Cursor { } // the character the cursor sits on; line terminators / EOF read as '\n'. -fn charAt(lines: []const []const u8, c: Cursor) u8 { +fn codepointAtCursor(lines: []const []const u8, c: Cursor) u21 { if (c.row >= lines.len) return '\n'; const line = lines[c.row]; if (c.col >= line.len) return '\n'; - return line[c.col]; + return codepointAt(line, graphemeStart(line, c.col)); +} + +fn charAt(lines: []const []const u8, c: Cursor) u8 { + const cp = codepointAtCursor(lines, c); + return if (cp <= 0x7f) @intCast(cp) else 0; } // `f`/`F`/`t`/`T`: the nth occurrence of `ch` after/before the cursor, across // line boundaries (helix: not confined to the line). `till` stops one position // short of the hit. Returns null (no move) when there aren't n occurrences. pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: bool, n: usize) ?Cursor { - if (ch > 0x7f) return null; // ponytail: ASCII targets only (byte columns) - const target: u8 = @intCast(ch); var p = clampToChar(lines, c); var left = if (n == 0) 1 else n; while (left > 0) { @@ -226,7 +270,7 @@ pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: } else { if (!stepBwd(lines, &p)) return null; } - if (charAt(lines, p) == target) left -= 1; + if (codepointAtCursor(lines, p) == ch) left -= 1; } if (till) { if (fwd) _ = stepBwd(lines, &p) else _ = stepFwd(lines, &p); @@ -369,16 +413,32 @@ pub fn wordRange(lines: []const []const u8, c0: Cursor, long: bool, around: bool const k = kindAt(lines, c, long); if (k == .ws) return null; var lo = c.col; - while (lo > 0 and kindAt(lines, .{ .row = c.row, .col = lo - 1 }, long) == k) lo -= 1; + while (lo > 0) { + const prev = prevGrapheme(line, lo); + if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != k) break; + lo = prev; + } var hi = c.col; - while (hi + 1 < line.len and kindAt(lines, .{ .row = c.row, .col = hi + 1 }, long) == k) hi += 1; + while (true) { + const next = nextGrapheme(line, hi); + if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != k) break; + hi = next; + } if (around) { var h2 = hi; - while (h2 + 1 < line.len and (line[h2 + 1] == ' ' or line[h2 + 1] == '\t')) h2 += 1; + while (true) { + const next = nextGrapheme(line, h2); + if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != .ws) break; + h2 = next; + } if (h2 != hi) { hi = h2; } else { - while (lo > 0 and (line[lo - 1] == ' ' or line[lo - 1] == '\t')) lo -= 1; + while (lo > 0) { + const prev = prevGrapheme(line, lo); + if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != .ws) break; + lo = prev; + } } } return .{ .a = .{ .row = c.row, .col = lo }, .b = .{ .row = c.row, .col = hi } }; @@ -403,7 +463,7 @@ pub fn paragraphRange(lines: []const []const u8, c0: Cursor, around: bool) ?Rang } } const llen = lineLenOf(lines, r1); - return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else llen - 1 } }; + return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else prevGrapheme(lines[r1], llen) } }; } // ---- helpers used by motions + main.zig ---- @@ -415,7 +475,7 @@ pub fn clampToChar(lines: []const []const u8, c: Cursor) Cursor { } const llen = lineLenOf(lines, c.row); if (llen == 0) return .{ .row = c.row, .col = 0 }; - return .{ .row = c.row, .col = @min(c.col, llen - 1) }; + return .{ .row = c.row, .col = clampLineCol(lines[c.row], c.col) }; } pub fn lineCount(content: []const u8) usize { @@ -465,8 +525,9 @@ pub fn insertAt(alloc: std.mem.Allocator, content: []const u8, c: Cursor, text: pub fn deleteChar(alloc: std.mem.Allocator, content: []const u8, c: Cursor) ![]u8 { const line = lineSlice(content, c.row); if (c.col >= line.len) return alloc.dupe(u8, content); - const off = lineStartOffset(content, c.row) + c.col; - return spliceAlloc(alloc, content, off, off + 1, ""); + const col = graphemeStart(line, c.col); + const line_start = lineStartOffset(content, c.row); + return spliceAlloc(alloc, content, line_start + col, line_start + nextGrapheme(line, col), ""); } // delete whole lines [r0, r1] inclusive (the line content + their terminators). @@ -520,9 +581,13 @@ fn rangeBytes(content: []const u8, a: Cursor, b: Cursor) struct { s: usize, e: u lo = b; hi = a; } - const s = lineStartOffset(content, lo.row) + @min(lo.col, lineSlice(content, lo.row).len); - var e = lineStartOffset(content, hi.row) + @min(hi.col, lineSlice(content, hi.row).len); - if (e < content.len) e += 1; // include the char under the head + const lo_line = lineSlice(content, lo.row); + const hi_line = lineSlice(content, hi.row); + const lo_col = graphemeStart(lo_line, @min(lo.col, lo_line.len)); + const hi_col = graphemeStart(hi_line, @min(hi.col, hi_line.len)); + const s = lineStartOffset(content, lo.row) + lo_col; + var e = lineStartOffset(content, hi.row) + hi_col; + if (e < content.len) e = nextGrapheme(content, e); // include the grapheme under the head return .{ .s = s, .e = @max(s, e) }; } @@ -593,10 +658,22 @@ pub fn replaceRange(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: // `r<ch>`: overwrite every char in the inclusive range [a, b] with `ch` β // NEWLINES TOO (helix replace maps every grapheme, so `xrz` joins lines). -pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u8) ![]u8 { +pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u21) ![]u8 { const r = rangeBytes(content, a, b); - const out = try alloc.dupe(u8, content); - for (out[r.s..r.e]) |*p| p.* = ch; + var encoded: [4]u8 = undefined; + const encoded_len = try std.unicode.utf8Encode(ch, &encoded); + var count: usize = 0; + var at = r.s; + while (at < r.e) : (count += 1) at = nextGrapheme(content, at); + const replacement_len = count * encoded_len; + const out = try alloc.alloc(u8, content.len - (r.e - r.s) + replacement_len); + @memcpy(out[0..r.s], content[0..r.s]); + var write = r.s; + for (0..count) |_| { + @memcpy(out[write..][0..encoded_len], encoded[0..encoded_len]); + write += encoded_len; + } + @memcpy(out[write..], content[r.e..]); return out; } @@ -729,9 +806,9 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C // // Gap-offset ranges over the FLAT buffer, ported faithfully from // helix-core/src/movement.rs + selection.rs @ 278b24389 (the genizah -// checkout). Positions are gap offsets 0..=text.len β "char indices" in -// helix terms, bytes here (ASCII-exact; UTF-8 stepped by sequence, matching -// the byte-column convention of the rest of pardes). A range with +// checkout). Positions are UTF-8 gap offsets 0..=text.len. Stored columns +// remain byte offsets, but every range boundary is an extended-grapheme +// boundary. A range with // head > anchor selects [anchor, head) with the block cursor ON head-1; // head < anchor selects [head, anchor) with the cursor ON head. The // differential suite (test/hxcases, `zig build hxdiff`) pins every behavior @@ -739,19 +816,58 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C pub const HxRange = struct { anchor: usize, head: usize }; -/// one grapheme forward (UTF-8 sequence step), clamped at text.len +/// The grapheme containing `off`, or text.len at EOF. This is also the repair +/// path for stale/external byte columns that happen to point into UTF-8. +pub fn graphemeStart(text: []const u8, off: usize) usize { + const bounded = @min(off, text.len); + if (bounded == text.len) return text.len; + var it = uucode.grapheme.utf8Iterator(text); + while (it.nextGrapheme()) |g| { + if (bounded < g.end) return g.start; + } + return text.len; +} + +/// one extended grapheme forward, clamped at text.len pub fn nextGrapheme(text: []const u8, off: usize) usize { if (off >= text.len) return text.len; - var o = off + 1; - while (o < text.len and (text[o] & 0xC0) == 0x80) o += 1; - return o; + // The editor's own offsets are already boundaries. Keep the overwhelmingly + // common ASCII path O(1); only repair a continuation-byte input here. + if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1; + var start = off; + while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1; + if (start != off) start = graphemeStart(text, off); + var it = uucode.grapheme.utf8Iterator(text[start..]); + const g = it.nextGrapheme() orelse return @min(start + 1, text.len); + return start + g.end; } pub fn prevGrapheme(text: []const u8, off: usize) usize { - if (off == 0) return 0; - var o = off - 1; - while (o > 0 and (text[o] & 0xC0) == 0x80) o -= 1; - return o; + var bounded = @min(off, text.len); + if (bounded < text.len) { + const repaired = graphemeStart(text, bounded); + if (repaired < bounded) return repaired; + bounded = repaired; + } + if (bounded == 0) return 0; + if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1; + // Graphemes cannot cross a line break. Restrict the forward segmentation + // needed for a reverse step to the current line instead of rescanning the + // complete buffer. + const line_start = if (std.mem.lastIndexOfScalar(u8, text[0 .. bounded - 1], '\n')) |nl| nl + 1 else 0; + if (line_start == bounded) return bounded - 1; // the newline is its own editor cell + var it = uucode.grapheme.utf8Iterator(text[line_start..bounded]); + var last = line_start; + while (it.nextGrapheme()) |g| last = line_start + g.start; + return last; +} + +/// Byte offset of the zero-based grapheme column, clamped to the line end. +pub fn graphemeAtColumn(text: []const u8, column: usize) usize { + var at: usize = 0; + var col: usize = 0; + while (at < text.len and col < column) : (col += 1) at = nextGrapheme(text, at); + return at; } /// the block cursor cell of a range (helix Range::cursor) @@ -792,14 +908,16 @@ pub fn hxLineEndIdx(text: []const u8, line: usize) usize { /// gap offset -> (row, col) cell pub fn hxPos(text: []const u8, off: usize) Cursor { - const o = @min(off, text.len); + const bounded = @min(off, text.len); // The line start is the byte after the last '\n' BEFORE off, which is the // same number lineStartOffset(text, row) walks the whole prefix to reach β // one backward scan of a single line instead of a second pass over // everything above the cursor. On a multi-MB buffer that second pass was // most of what a keystroke cost. - const s = if (std.mem.lastIndexOfScalar(u8, text[0..o], '\n')) |nl| nl + 1 else 0; - return .{ .row = hxLineOf(text, off), .col = o - s }; + const s = if (std.mem.lastIndexOfScalar(u8, text[0..bounded], '\n')) |nl| nl + 1 else 0; + const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; + const col = graphemeStart(text[s..e], @min(bounded - s, e - s)); + return .{ .row = hxLineOf(text, s + col), .col = col }; } /// (row, col) -> clamped gap offset; col == line length lands ON the '\n' @@ -809,7 +927,8 @@ pub fn hxOff(text: []const u8, c: Cursor) usize { // hxLineEndIdx(text, row) inlined: it starts by walking to `row` again, // and we are already standing there const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; - return @min(s + c.col, e); + const raw = @min(c.col, e - s); + return s + graphemeStart(text[s..e], raw); } pub const WordTarget = enum { @@ -827,35 +946,36 @@ pub const WordTarget = enum { // that distinction is load-bearing in reached_target. const HxCat = enum { word, punct, ws, eol }; -fn hxCat(b: u8) HxCat { - if (b == '\n' or b == '\r') return .eol; - if (b == ' ' or b == '\t' or b == 0x0b or b == 0x0c) return .ws; - if (std.ascii.isAlphanumeric(b) or b == '_' or b >= 0x80) return .word; - return .punct; +fn hxCatAt(text: []const u8, off: usize) HxCat { + if (off >= text.len) return .eol; + const cp = codepointAt(text, off); + if (cp == '\n' or cp == '\r') return .eol; + return switch (kindOfCodepoint(cp)) { + .word => .word, + .punct => .punct, + .ws => .ws, + }; } -fn hxIsWs(b: u8) bool { // Rust char::is_whitespace (includes line endings) - const c = hxCat(b); +fn hxIsWs(c: HxCat) bool { // Rust char::is_whitespace (includes line endings) return c == .ws or c == .eol; } -fn hxIsWordBoundary(a: u8, b: u8) bool { - return hxCat(a) != hxCat(b); +fn hxIsWordBoundary(a: HxCat, b: HxCat) bool { + return a != b; } -fn hxIsLongBoundary(a: u8, b: u8) bool { - const ca = hxCat(a); - const cb = hxCat(b); - if ((ca == .word and cb == .punct) or (ca == .punct and cb == .word)) return false; - return ca != cb; +fn hxIsLongBoundary(a: HxCat, b: HxCat) bool { + if ((a == .word and b == .punct) or (a == .punct and b == .word)) return false; + return a != b; } -fn hxReached(target: WordTarget, prev: u8, next: u8) bool { +fn hxReached(target: WordTarget, prev: HxCat, next: HxCat) bool { return switch (target) { - .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)), - .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol), - .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)), - .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol), + .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (next == .eol or !hxIsWs(next)), + .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or next == .eol), + .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (next == .eol or !hxIsWs(next)), + .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or next == .eol), }; } @@ -896,45 +1016,35 @@ fn hxRangeToTarget(text: []const u8, target: WordTarget, origin: HxRange, is_pre var anchor = origin.anchor; var head = origin.head; var it = origin.head; - var prev_ch: ?u8 = if (is_prev) - (if (it < text.len) text[it] else null) + var prev_cat: ?HxCat = if (is_prev) + (if (it < text.len) hxCatAt(text, it) else null) else - (if (it > 0) text[it - 1] else null); + (if (it > 0) hxCatAt(text, prevGrapheme(text, it)) else null); // skip any initial newline characters while (true) { - const ch: u8 = if (is_prev) blk: { - if (it == 0) break; - break :blk text[it - 1]; - } else blk: { - if (it >= text.len) break; - break :blk text[it]; - }; - if (ch != '\n' and ch != '\r') break; - if (is_prev) it -= 1 else it += 1; - prev_ch = ch; - if (is_prev) head -|= 1 else head += 1; + if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break; + const cell = if (is_prev) prevGrapheme(text, it) else it; + const cat = hxCatAt(text, cell); + if (cat != .eol) break; + it = if (is_prev) cell else nextGrapheme(text, cell); + prev_cat = cat; + head = it; } - if (prev_ch != null and hxCat(prev_ch.?) == .eol) anchor = head; + if (prev_cat == .eol) anchor = head; // find the target position const head_start = head; while (true) { - const next_ch: u8 = if (is_prev) blk: { - if (it == 0) break; - it -= 1; - break :blk text[it]; - } else blk: { - if (it >= text.len) break; - const c = text[it]; - it += 1; - break :blk c; - }; - if (prev_ch == null or hxReached(target, prev_ch.?, next_ch)) { + if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break; + const cell = if (is_prev) prevGrapheme(text, it) else it; + const next_cat = hxCatAt(text, cell); + if (prev_cat == null or hxReached(target, prev_cat.?, next_cat)) { if (head == head_start) anchor = head else break; } - prev_ch = next_ch; - if (is_prev) head -|= 1 else head += 1; + prev_cat = next_cat; + it = if (is_prev) cell else nextGrapheme(text, cell); + head = it; } return .{ .anchor = anchor, .head = head }; } @@ -1007,31 +1117,31 @@ pub fn hxVertTarget(text: []const u8, pos: usize, down: bool, count: usize, goal const s = lineStartOffset(text, nline); // hxLineEndIdx(text, nline) without its second walk to nline (see hxOff) const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; - return @min(s + goal_col, e); + return s + graphemeStart(text[s..e], @min(goal_col, e - s)); } /// f/F/t/T target cell. helix find_char: the exclusive (till) search starts /// one further out so repeats make progress; not-found = null (no move). -pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bool, count: usize) ?usize { +pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u21, fwd: bool, till: bool, count: usize) ?usize { var left = @max(1, count); if (fwd) { const head = nextGrapheme(text, cursor); - var i = if (till) head + 1 else head; + var i = if (till) nextGrapheme(text, head) else head; if (i > text.len) return null; - while (i < text.len) : (i += 1) { - if (text[i] == ch) { + while (i < text.len) : (i = nextGrapheme(text, i)) { + if (codepointAt(text, i) == ch) { left -= 1; - if (left == 0) return if (till) i - 1 else i; + if (left == 0) return if (till) prevGrapheme(text, i) else i; } } return null; } - var i = if (till) cursor -| 1 else cursor; + var i = if (till) prevGrapheme(text, cursor) else cursor; while (i > 0) { - i -= 1; - if (text[i] == ch) { + i = prevGrapheme(text, i); + if (codepointAt(text, i) == ch) { left -= 1; - if (left == 0) return if (till) i + 1 else i; + if (left == 0) return if (till) nextGrapheme(text, i) else i; } } return null; @@ -1040,26 +1150,19 @@ pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bo // helix textobject.rs find_word_boundary fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usize { var prev: HxCat = if (fwd) - (if (pos0 == 0) .ws else hxCat(text[pos0 - 1])) + (if (pos0 == 0) .ws else hxCatAt(text, prevGrapheme(text, pos0))) else - (if (pos0 >= text.len) .ws else hxCat(text[pos0])); + (if (pos0 >= text.len) .ws else hxCatAt(text, pos0)); var pos = pos0; var it = pos0; while (true) { - const ch: u8 = if (fwd) blk: { - if (it >= text.len) break; - const c = text[it]; - it += 1; - break :blk c; - } else blk: { - if (it == 0) break; - it -= 1; - break :blk text[it]; - }; - const cat = hxCat(ch); + if ((fwd and it >= text.len) or (!fwd and it == 0)) break; + const cell = if (fwd) it else prevGrapheme(text, it); + const cat = hxCatAt(text, cell); if (cat == .eol or cat == .ws) return pos; if (!long and cat != prev and pos != 0 and pos != text.len) return pos; - if (fwd) pos += 1 else pos -|= 1; + it = if (fwd) nextGrapheme(text, cell) else cell; + pos = it; prev = cat; } return pos; @@ -1070,14 +1173,20 @@ fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usiz pub fn hxTextobjectWord(text: []const u8, r: HxRange, around: bool, long: bool) HxRange { const pos = hxCursor(text, r); const word_start = hxFindWordBoundary(text, pos, false, long); - const cat: HxCat = if (pos < text.len) hxCat(text[pos]) else .ws; - const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, pos + 1, true, long); + const cat: HxCat = if (pos < text.len) hxCatAt(text, pos) else .ws; + const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, nextGrapheme(text, pos), true, long); if (word_start == word_end or !around) return .{ .anchor = word_start, .head = word_end }; var end = word_end; - while (end < text.len and hxIsWs(text[end]) and hxCat(text[end]) != .eol) end += 1; + while (end < text.len and hxIsWs(hxCatAt(text, end)) and hxCatAt(text, end) != .eol) + end = nextGrapheme(text, end); if (end > word_end) return .{ .anchor = word_start, .head = end }; var start = word_start; - while (start > 0 and hxIsWs(text[start - 1]) and hxCat(text[start - 1]) != .eol) start -= 1; + while (start > 0) { + const before = prevGrapheme(text, start); + const before_cat = hxCatAt(text, before); + if (!hxIsWs(before_cat) or before_cat == .eol) break; + start = before; + } return .{ .anchor = start, .head = word_end }; } @@ -1294,6 +1403,38 @@ test "hx find targets" { try std.testing.expectEqual(@as(?usize, null), hxFindTarget(t, 0, 'z', true, false, 1)); } +test "extended grapheme boundaries cover combining emoji flag and CJK text" { + const text = "a" ++ "e\u{301}" ++ "π©π½\u{200d}π" ++ "π§π·" ++ "η"; + const boundaries = [_]usize{ 0, 1, 4, 19, 27, 30 }; + for (boundaries[0 .. boundaries.len - 1], boundaries[1..]) |start, end| { + try std.testing.expectEqual(end, nextGrapheme(text, start)); + try std.testing.expectEqual(start, prevGrapheme(text, end)); + } + // Stale byte offsets are repaired to a cluster boundary instead of being + // allowed to leak continuation bytes into cursor state. + try std.testing.expectEqual(@as(usize, 1), graphemeStart(text, 2)); + try std.testing.expectEqual(@as(usize, 1), prevGrapheme(text, 3)); + try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3)); +} + +test "Unicode find and word motion stay on grapheme boundaries" { + const lines = [_][]const u8{"\u{e9}x\u{e9}"}; + try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'Γ©', true, false, 1).?); + + const text = "cafΓ© δΈη ok\n"; + const first = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 1, .next_word_start); + try std.testing.expectEqual(@as(usize, 0), first.anchor); + try std.testing.expectEqual(@as(usize, 6), first.head); + const second = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 2, .next_word_start); + try std.testing.expectEqual(@as(usize, 6), second.anchor); + try std.testing.expectEqual(@as(usize, 13), second.head); + + // Long-word motions split on Unicode whitespace, not only ASCII spaces. + const nbsp = "alpha\u{a0}beta\n"; + const long = hxWordMove(nbsp, .{ .anchor = 0, .head = 1 }, 1, .next_long_word_start); + try std.testing.expectEqual(@as(usize, 7), long.head); +} + test "hx increment" { const a = std.testing.allocator; { @@ -1335,7 +1476,7 @@ test "kindOf" { test "char/line motions" { const lines = [_][]const u8{ "alpha beta", " two words", "x" }; const c = Cursor{ .row = 0, .col = 5 }; - try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(c)); + try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(&.{"hello"}, c)); try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, charRight(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 1, .col = 5 }, lineDown(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, lineUp(&lines, Cursor{ .row = 1, .col = 5 })); @@ -1440,6 +1581,19 @@ test "deleteChar" { try std.testing.expectEqualStrings("abc", r2); } +test "Unicode edits replace and delete whole graphemes" { + const a = std.testing.allocator; + const content = "A" ++ "e\u{301}" ++ "π©π½\u{200d}π" ++ "π§π·" ++ "η" ++ "Z"; + + const deleted = try deleteChar(a, content, .{ .row = 0, .col = 2 }); + defer a.free(deleted); + try std.testing.expectEqualStrings("Aπ©π½\u{200d}ππ§π·ηZ", deleted); + + const replaced = try replaceChars(a, content, .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 19 }, 'η'); + defer a.free(replaced); + try std.testing.expectEqualStrings("AηηηηZ", replaced); +} + test "deleteLines middle" { const content = "one\ntwo\nthree\nfour"; const a = std.testing.allocator; |
