const std = @import("std"); const uucode = @import("uucode"); // Modal-editing text math, kept free of vaxis/ghostty so it can be unit-tested // in isolation (see the `unit-test` build step). main.zig wires this onto the // pane's cursor + (for file panes) its content. // // The cursor sits ON a grapheme: col is its UTF-8 byte offset in // [0, line.len]; col == line.len means "on the line terminator / after the // last grapheme". Motions never leave a cursor in the middle of UTF-8 or an // extended grapheme cluster. pub const Cursor = struct { row: usize = 0, col: usize = 0, pub fn eql(a: Cursor, b: Cursor) bool { return a.row == b.row and a.col == b.col; } }; // word char classes (matches ad/vim/kakoune: word = alnum + _, punct = other // non-ws, ws = space/tab/newline). pub const Kind = enum { word, punct, ws }; fn codepointAt(text: []const u8, off: usize) u21 { if (off >= text.len) return 0xFFFD; const n = std.unicode.utf8ByteSequenceLength(text[off]) catch return 0xFFFD; if (off + n > text.len) return 0xFFFD; return std.unicode.utf8Decode(text[off .. off + n]) catch 0xFFFD; } fn isUnicodeWhitespace(cp: u21) bool { if (cp == ' ' or (cp >= '\t' and cp <= '\r') or cp == 0x85) return true; return switch (uucode.get(.general_category, cp)) { .separator_space, .separator_line, .separator_paragraph => true, else => false, }; } fn kindOfCodepoint(cp: u21) Kind { if (isUnicodeWhitespace(cp)) return .ws; if (cp == '_') return .word; return switch (uucode.get(.general_category, cp)) { .letter_uppercase, .letter_lowercase, .letter_titlecase, .letter_modifier, .letter_other, .mark_nonspacing, .mark_spacing_combining, .mark_enclosing, .number_decimal_digit, .number_letter, .number_other, .punctuation_connector, => .word, else => .punct, }; } pub fn kindOf(c: u8) Kind { return kindOfCodepoint(c); } // "long word" (W/B/E): only whitespace separates; punct is part of a word. fn kindOfLong(cp: u21) Kind { return if (isUnicodeWhitespace(cp)) .ws else .word; } fn kindAt(lines: []const []const u8, c: Cursor, long: bool) Kind { if (c.row >= lines.len) return .ws; const line = lines[c.row]; if (c.col >= line.len) return .ws; // line terminator / EOF = whitespace const cp = codepointAt(line, graphemeStart(line, c.col)); return if (long) kindOfLong(cp) else kindOfCodepoint(cp); } fn lineLenOf(lines: []const []const u8, row: usize) usize { if (row >= lines.len) return 0; return lines[row].len; } // advance one position across line boundaries (line terminators are positions // too: col == line.len is the newline). Returns false at EOF. fn stepFwd(lines: []const []const u8, c: *Cursor) bool { if (c.row >= lines.len) return false; const llen = lineLenOf(lines, c.row); if (c.col < llen) { c.col = nextGrapheme(lines[c.row], c.col); return true; } // at the newline: move to next line start if (c.row + 1 < lines.len) { c.row += 1; c.col = 0; return true; } return false; // EOF } fn stepBwd(lines: []const []const u8, c: *Cursor) bool { if (c.col > 0) { c.col = prevGrapheme(lines[c.row], c.col); return true; } if (c.row == 0) return false; c.row -= 1; c.col = lineLenOf(lines, c.row); // the previous line's newline return true; } // at EOF? (past the last line's last char) fn atEof(lines: []const []const u8, c: Cursor) bool { if (c.row >= lines.len) return true; if (c.row + 1 < lines.len) return false; return c.col >= lines[c.row].len; } pub fn firstNonWs(line: []const u8) usize { var i: usize = 0; while (i < line.len and isUnicodeWhitespace(codepointAt(line, i))) i = nextGrapheme(line, i); return i; } // ---- per-line motions ---- pub fn lineStart(c: Cursor) Cursor { return .{ .row = c.row, .col = 0 }; } pub fn lineEnd(lines: []const []const u8, c: Cursor) Cursor { const llen = lineLenOf(lines, c.row); return .{ .row = c.row, .col = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen) }; } pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor { // the row can sit past the content (mouse click below a short pane's // last line) — out of range reads as an empty line, like lineLenOf if (c.row >= lines.len) return .{ .row = c.row, .col = 0 }; return .{ .row = c.row, .col = firstNonWs(lines[c.row]) }; } // ---- char/line motions ---- pub fn charLeft(lines: []const []const u8, c: Cursor) Cursor { if (c.row >= lines.len) return .{ .row = c.row, .col = 0 }; return .{ .row = c.row, .col = prevGrapheme(lines[c.row], @min(c.col, lines[c.row].len)) }; } pub fn charRight(lines: []const []const u8, c: Cursor) Cursor { const llen = lineLenOf(lines, c.row); const last = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen); return .{ .row = c.row, .col = if (c.col < last) @min(nextGrapheme(lines[c.row], c.col), last) else last }; } fn clampLineCol(line: []const u8, col: usize) usize { if (line.len == 0) return 0; const last = prevGrapheme(line, line.len); return graphemeStart(line, @min(col, last)); } pub fn lineDown(lines: []const []const u8, c: Cursor) Cursor { const nr = if (c.row + 1 < lines.len) c.row + 1 else c.row; const llen = lineLenOf(lines, nr); return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) }; } pub fn lineUp(lines: []const []const u8, c: Cursor) Cursor { const nr = if (c.row > 0) c.row - 1 else c.row; const llen = lineLenOf(lines, nr); return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) }; } // ---- word motions ---- // `w`/`W`: to the start of the next word. pub fn nextWordStart(lines: []const []const u8, c: Cursor, long: bool) Cursor { var p = c; const start_kind = kindAt(lines, p, long); if (start_kind != .ws) { // skip the rest of the current word-class run while (!atEof(lines, p) and kindAt(lines, p, long) == start_kind) { if (!stepFwd(lines, &p)) break; } } // skip whitespace (incl. newlines) to the next word start while (!atEof(lines, p) and kindAt(lines, p, long) == .ws) { if (!stepFwd(lines, &p)) break; } // p now sits on the next word's first char (or EOF -> last valid pos) return clampToChar(lines, p); } // `b`/`B`: to the start of the previous word. pub fn prevWordStart(lines: []const []const u8, c: Cursor, long: bool) Cursor { var p = c; if (!stepBwd(lines, &p)) return c; // at buffer start // skip whitespace backward while (kindAt(lines, p, long) == .ws) { if (!stepBwd(lines, &p)) return .{ .row = 0, .col = 0 }; } // now on the end of the previous word; walk back to its start const k = kindAt(lines, p, long); while (true) { var q = p; if (!stepBwd(lines, &q)) { p.col = 0; break; } if (kindAt(lines, q, long) != k) break; // crossed into prior class p = q; } return clampToChar(lines, p); } // `e`/`E`: to the end of the current/next word. pub fn nextWordEnd(lines: []const []const u8, c: Cursor, long: bool) Cursor { var p = c; if (!stepFwd(lines, &p)) return clampToChar(lines, c); // skip whitespace forward while (!atEof(lines, p) and kindAt(lines, p, long) == .ws) { if (!stepFwd(lines, &p)) break; } if (atEof(lines, p)) return clampToChar(lines, p); // now on a word's first char; advance to the last char of this run const k = kindAt(lines, p, long); while (!atEof(lines, p)) { var q = p; if (!stepFwd(lines, &q)) break; if (kindAt(lines, q, long) != k) break; p = q; } return clampToChar(lines, p); } // ---- goto ---- pub fn gotoFirst() Cursor { return .{ .row = 0, .col = 0 }; } pub fn gotoLast(lines: []const []const u8) Cursor { const r = if (lines.len == 0) 0 else lines.len - 1; return .{ .row = r, .col = 0 }; } // the character the cursor sits on; line terminators / EOF read as '\n'. fn codepointAtCursor(lines: []const []const u8, c: Cursor) u21 { if (c.row >= lines.len) return '\n'; const line = lines[c.row]; if (c.col >= line.len) return '\n'; return codepointAt(line, graphemeStart(line, c.col)); } fn charAt(lines: []const []const u8, c: Cursor) u8 { const cp = codepointAtCursor(lines, c); return if (cp <= 0x7f) @intCast(cp) else 0; } // `f`/`F`/`t`/`T`: the nth occurrence of `ch` after/before the cursor, across // line boundaries (helix: not confined to the line). `till` stops one position // short of the hit. Returns null (no move) when there aren't n occurrences. pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: bool, n: usize) ?Cursor { var p = clampToChar(lines, c); var left = if (n == 0) 1 else n; while (left > 0) { if (fwd) { if (!stepFwd(lines, &p)) return null; } else { if (!stepBwd(lines, &p)) return null; } if (codepointAtCursor(lines, p) == ch) left -= 1; } if (till) { if (fwd) _ = stepBwd(lines, &p) else _ = stepFwd(lines, &p); } return clampToChar(lines, p); } // `mm`: the bracket matching the one under the cursor (dumb text scan with // nesting; no tree-sitter). Null when the cursor is not on a bracket. pub fn matchBracket(lines: []const []const u8, c: Cursor) ?Cursor { const opens = "([{<"; const closes = ")]}>"; const start = clampToChar(lines, c); const ch = charAt(lines, start); if (std.mem.indexOfScalar(u8, opens, ch)) |i| { var depth: usize = 0; var p = start; while (true) { const cc = charAt(lines, p); if (cc == opens[i]) depth += 1; if (cc == closes[i]) { depth -= 1; if (depth == 0) return p; } if (!stepFwd(lines, &p)) return null; } } if (std.mem.indexOfScalar(u8, closes, ch)) |i| { var depth: usize = 0; var p = start; while (true) { const cc = charAt(lines, p); if (cc == closes[i]) depth += 1; if (cc == opens[i]) { depth -= 1; if (depth == 0) return p; } if (!stepBwd(lines, &p)) return null; } } return null; } fn isBlank(line: []const u8) bool { return firstNonWs(line) == line.len; } // `]p`: the start of the next blank-line-delimited block (or the last line). pub fn paragraphFwd(lines: []const []const u8, c: Cursor) Cursor { var r = c.row; while (r < lines.len and !isBlank(lines[r])) r += 1; while (r < lines.len and isBlank(lines[r])) r += 1; if (r >= lines.len) return gotoLast(lines); return .{ .row = r, .col = 0 }; } // `[p`: the start of the current block, or of the previous one when already // on a block start / a blank line. pub fn paragraphBwd(lines: []const []const u8, c: Cursor) Cursor { if (c.row == 0 or lines.len == 0) return .{ .row = 0, .col = 0 }; var r = @min(c.row, lines.len) - 1; while (r > 0 and isBlank(lines[r])) r -= 1; while (r > 0 and !isBlank(lines[r - 1])) r -= 1; return .{ .row = r, .col = 0 }; } // ---- textobject / surround range math (mi/ma/ms/mr/md) ---- // an inclusive char range [a, b] in document order pub const Range = struct { a: Cursor, b: Cursor }; // the nearest pair of brackets enclosing the cursor (nesting-aware; the // cursor sitting ON a bracket belongs to that pair). Positions of the // bracket chars themselves. pub fn enclosingPair(lines: []const []const u8, c: Cursor, open: u8, close: u8) ?Range { var a = clampToChar(lines, c); if (charAt(lines, a) != open) { var depth: usize = 0; while (true) { if (!stepBwd(lines, &a)) return null; const ch = charAt(lines, a); if (ch == close) depth += 1; if (ch == open) { if (depth == 0) break; depth -= 1; } } } const b = matchBracket(lines, a) orelse return null; return .{ .a = a, .b = b }; } // the quote pair around the cursor, scanned on the cursor's line only // (plain-text strings don't span lines). Positions of the quote chars. pub fn enclosingQuote(lines: []const []const u8, c0: Cursor, q: u8) ?Range { const c = clampToChar(lines, c0); if (c.row >= lines.len) return null; const line = lines[c.row]; var i: usize = 0; while (i < line.len) { const o = std.mem.indexOfScalarPos(u8, line, i, q) orelse return null; const e = std.mem.indexOfScalarPos(u8, line, o + 1, q) orelse return null; if (c.col < o) return null; // the cursor sits before any pair if (c.col <= e) return .{ .a = .{ .row = c.row, .col = o }, .b = .{ .row = c.row, .col = e } }; i = e + 1; } return null; } // mi/ma over a bracket pair: `around` keeps the brackets, inside shrinks them // off (null when nothing is left between them). pub fn pairRange(lines: []const []const u8, c: Cursor, open: u8, close: u8, around: bool) ?Range { const r = enclosingPair(lines, c, open, close) orelse return null; if (around) return r; return shrinkOffDelims(lines, r); } pub fn quoteRange(lines: []const []const u8, c: Cursor, q: u8, around: bool) ?Range { const r = enclosingQuote(lines, c, q) orelse return null; if (around) return r; return shrinkOffDelims(lines, r); } fn shrinkOffDelims(lines: []const []const u8, r: Range) ?Range { var a = r.a; var b = r.b; if (!stepFwd(lines, &a)) return null; if (!stepBwd(lines, &b)) return null; if (b.row < a.row or (b.row == a.row and b.col < a.col)) return null; // empty inside return .{ .a = a, .b = b }; } // miw/maw (and W): the word run under the cursor; `around` adds the trailing // whitespace on the line (or the leading run when there is none). pub fn wordRange(lines: []const []const u8, c0: Cursor, long: bool, around: bool) ?Range { const c = clampToChar(lines, c0); if (c.row >= lines.len) return null; const line = lines[c.row]; if (line.len == 0 or c.col >= line.len) return null; const k = kindAt(lines, c, long); if (k == .ws) return null; var lo = c.col; while (lo > 0) { const prev = prevGrapheme(line, lo); if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != k) break; lo = prev; } var hi = c.col; while (true) { const next = nextGrapheme(line, hi); if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != k) break; hi = next; } if (around) { var h2 = hi; while (true) { const next = nextGrapheme(line, h2); if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != .ws) break; h2 = next; } if (h2 != hi) { hi = h2; } else { while (lo > 0) { const prev = prevGrapheme(line, lo); if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != .ws) break; lo = prev; } } } return .{ .a = .{ .row = c.row, .col = lo }, .b = .{ .row = c.row, .col = hi } }; } // mip/map: the blank-line-delimited block around the cursor; `around` adds the // trailing blank lines (or the leading ones when there are none). pub fn paragraphRange(lines: []const []const u8, c0: Cursor, around: bool) ?Range { const c = clampToChar(lines, c0); if (c.row >= lines.len or isBlank(lines[c.row])) return null; var r0 = c.row; while (r0 > 0 and !isBlank(lines[r0 - 1])) r0 -= 1; var r1 = c.row; while (r1 + 1 < lines.len and !isBlank(lines[r1 + 1])) r1 += 1; if (around) { var r2 = r1; while (r2 + 1 < lines.len and isBlank(lines[r2 + 1])) r2 += 1; if (r2 != r1) { r1 = r2; } else { while (r0 > 0 and isBlank(lines[r0 - 1])) r0 -= 1; } } const llen = lineLenOf(lines, r1); return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else prevGrapheme(lines[r1], llen) } }; } // ---- helpers used by motions + main.zig ---- // clamp a (possibly terminator/EOF) position onto a real character. pub fn clampToChar(lines: []const []const u8, c: Cursor) Cursor { if (c.row >= lines.len) { return .{ .row = if (lines.len == 0) 0 else lines.len - 1, .col = 0 }; } const llen = lineLenOf(lines, c.row); if (llen == 0) return .{ .row = c.row, .col = 0 }; return .{ .row = c.row, .col = clampLineCol(lines[c.row], c.col) }; } pub fn lineCount(content: []const u8) usize { if (content.len == 0) return 0; return std.mem.count(u8, content, "\n") + 1; } // byte offset of the start of line `row` (0-based). row may == lineCount() // (== content.len, the end). pub fn lineStartOffset(content: []const u8, row: usize) usize { var off: usize = 0; var r: usize = 0; while (r < row) : (r += 1) { const nl = std.mem.indexOfScalarPos(u8, content, off, '\n') orelse return content.len; off = nl + 1; } return off; } // the text of line `row` (no terminator), a slice into `content`. pub fn lineSlice(content: []const u8, row: usize) []const u8 { const start = lineStartOffset(content, row); if (start >= content.len) return ""; // indexOfScalarPos, not indexOfPos with a one-byte needle: the latter runs the generic // substring search where a memchr will do, and this is called once per visible row per frame. const nl = std.mem.indexOfScalarPos(u8, content, start, '\n') orelse content.len; return content[start..nl]; } /// The byte span of line `row`, in ONE scan that stops at that row. /// /// This exists because the obvious spelling costs a scan of the WHOLE document per call and the /// obvious USE of it costs several. `insertAt` below read `lineCount` twice merely to clamp a row, /// and `lineCount` is `std.mem.count` over every byte; on a 19 MB fixture that was two full passes /// before a single character could be inserted. Measured with `zig build perf`: `edit-char` on the /// 300 000-line fixture cost 15.0 ms, against 1.5 ms to render the frame that shows it. /// /// Returns null when the row does not exist, so a caller that must clamp pays for the count only on /// that path - which is the rare one, since a cursor is normally inside its document. pub const LineSpan = struct { start: usize, end: usize }; pub fn lineSpan(content: []const u8, row: usize) ?LineSpan { // An empty document has no lines at all, which is what `lineCount` says about it - not one // empty line. Agreeing with that here is what lets `insertAt` fall through to offset 0. if (content.len == 0) return null; var start: usize = 0; var r: usize = 0; while (r < row) : (r += 1) { const nl = std.mem.indexOfScalarPos(u8, content, start, '\n') orelse return null; start = nl + 1; } // Row `row` exists if it begins inside the content, OR it is the empty last line after a // trailing newline - which `lineCount` also counts, so the two agree. if (start > content.len) return null; if (start == content.len and !(row == 0 or content.len == 0 or content[content.len - 1] == '\n')) return null; const end = std.mem.indexOfScalarPos(u8, content, start, '\n') orelse content.len; return .{ .start = start, .end = end }; } // ---- file content mutations. caller frees the returned slice + the old one. ---- fn spliceAlloc(alloc: std.mem.Allocator, content: []const u8, start: usize, end: usize, replacement: []const u8) ![]u8 { const out = try alloc.alloc(u8, content.len - (end - start) + replacement.len); @memcpy(out[0..start], content[0..start]); @memcpy(out[start..][0..replacement.len], replacement); @memcpy(out[start + replacement.len ..], content[end..]); return out; } /// insert `text` at (row, col). col is clamped to the line length. pub fn insertAt(alloc: std.mem.Allocator, content: []const u8, c: Cursor, text: []const u8) ![]u8 { // One bounded scan on the common path. The fallback keeps the old clamping exactly - a row past // the end lands on the last line - and only it pays for a full count. const span = lineSpan(content, c.row) orelse blk: { const last = lineCount(content) -| 1; break :blk lineSpan(content, last) orelse LineSpan{ .start = content.len, .end = content.len }; }; const off = span.start + @min(c.col, span.end - span.start); return spliceAlloc(alloc, content, off, off, text); } // delete the character at (row, col). no-op if col is past the line end. pub fn deleteChar(alloc: std.mem.Allocator, content: []const u8, c: Cursor) ![]u8 { const line = lineSlice(content, c.row); if (c.col >= line.len) return alloc.dupe(u8, content); const col = graphemeStart(line, c.col); const line_start = lineStartOffset(content, c.row); return spliceAlloc(alloc, content, line_start + col, line_start + nextGrapheme(line, col), ""); } // delete whole lines [r0, r1] inclusive (the line content + their terminators). // returns the new content; `deleted` is the joined removed text (no terminators). pub const Deleted = struct { content: []u8, deleted: []u8 }; pub fn deleteLines(alloc: std.mem.Allocator, content: []const u8, r0: usize, r1: usize) !Deleted { const n = lineCount(content); const lo = @min(r0, r1); const hi = @min(@max(r0, r1), if (n == 0) 0 else n - 1); if (n == 0 or hi < lo) { const unchanged = try alloc.dupe(u8, content); errdefer alloc.free(unchanged); return .{ .content = unchanged, .deleted = try alloc.dupe(u8, "") }; } const start = lineStartOffset(content, lo); // end = start of line (hi+1), or content.len if hi is the last line. const end = if (hi + 1 < n) lineStartOffset(content, hi + 1) else content.len; // if we're removing the last line and there's a preceding newline, also // drop that newline so we don't leave a trailing blank line. var cut_lo = start; const cut_hi = end; if (hi + 1 == n and start > 0) cut_lo -= 1; // remove the '\n' before the last line var deleted_len: usize = hi - lo; var r = lo; while (r <= hi) : (r += 1) deleted_len += lineSlice(content, r).len; const deleted = try alloc.alloc(u8, deleted_len); errdefer alloc.free(deleted); var write: usize = 0; r = lo; while (r <= hi) : (r += 1) { if (r > lo) { deleted[write] = '\n'; write += 1; } const line = lineSlice(content, r); @memcpy(deleted[write..][0..line.len], line); write += line.len; } return .{ .content = try spliceAlloc(alloc, content, cut_lo, cut_hi, ""), .deleted = deleted }; } // normalize two cursors into document order (lo <= hi), then byte offsets of the // INCLUSIVE range [lo .. hi] (the char under hi is included). e == s means empty. fn rangeBytes(content: []const u8, a: Cursor, b: Cursor) struct { s: usize, e: usize } { var lo = a; var hi = b; if (hi.row < lo.row or (hi.row == lo.row and hi.col < lo.col)) { lo = b; hi = a; } const lo_line = lineSlice(content, lo.row); const hi_line = lineSlice(content, hi.row); const lo_col = graphemeStart(lo_line, @min(lo.col, lo_line.len)); const hi_col = graphemeStart(hi_line, @min(hi.col, hi_line.len)); const s = lineStartOffset(content, lo.row) + lo_col; var e = lineStartOffset(content, hi.row) + hi_col; if (e < content.len) e = nextGrapheme(content, e); // include the grapheme under the head return .{ .s = s, .e = @max(s, e) }; } // the text of the inclusive char range [a, b] (cursors in either order). Caller frees. pub fn rangeText(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor) ![]u8 { const r = rangeBytes(content, a, b); return alloc.dupe(u8, content[r.s..r.e]); } // delete the inclusive char range [a, b]. returns new content + the removed text. pub fn deleteRange(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor) !Deleted { const r = rangeBytes(content, a, b); const deleted = try alloc.dupe(u8, content[r.s..r.e]); errdefer alloc.free(deleted); return .{ .content = try spliceAlloc(alloc, content, r.s, r.e, ""), .deleted = deleted }; } // replace line `row`'s text with "" (keep the line, empty it). For `c`hange line. pub fn clearLine(alloc: std.mem.Allocator, content: []const u8, row: usize) ![]u8 { const line = lineSlice(content, row); const start = lineStartOffset(content, row); return spliceAlloc(alloc, content, start, start + line.len, ""); } // paste `text` as a new line BELOW `row`. Multiline `text` becomes several lines. pub fn pasteLineBelow(alloc: std.mem.Allocator, content: []const u8, row: usize, text: []const u8) ![]u8 { const n = lineCount(content); const off = if (row + 1 < n) lineStartOffset(content, row + 1) else content.len; const suffix_newline = off < content.len or (content.len > 0 and content[content.len - 1] == '\n'); const prefix_newline = !suffix_newline and off > 0; const extra: usize = @intFromBool(suffix_newline or prefix_newline); const out = try alloc.alloc(u8, content.len + text.len + extra); var write: usize = 0; @memcpy(out[write..][0..off], content[0..off]); write += off; if (prefix_newline) { out[write] = '\n'; write += 1; } @memcpy(out[write..][0..text.len], text); write += text.len; if (suffix_newline) { out[write] = '\n'; write += 1; } @memcpy(out[write..], content[off..]); return out; } // the cursor position AFTER `text` inserted at `c` (one past its last char). pub fn advanceBy(c: Cursor, text: []const u8) Cursor { var r = c.row; var col = c.col; for (text) |ch| { if (ch == '\n') { r += 1; col = 0; } else col += 1; } return .{ .row = r, .col = col }; } // replace the inclusive char range [a, b] with `text` (R replace-with-yank). pub fn replaceRange(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, text: []const u8) ![]u8 { const r = rangeBytes(content, a, b); return spliceAlloc(alloc, content, r.s, r.e, text); } // `r`: overwrite every char in the inclusive range [a, b] with `ch` — // NEWLINES TOO (helix replace maps every grapheme, so `xrz` joins lines). pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u21) ![]u8 { const r = rangeBytes(content, a, b); var encoded: [4]u8 = undefined; const encoded_len = try std.unicode.utf8Encode(ch, &encoded); var count: usize = 0; var at = r.s; while (at < r.e) : (count += 1) at = nextGrapheme(content, at); const replacement_len = count * encoded_len; const out = try alloc.alloc(u8, content.len - (r.e - r.s) + replacement_len); @memcpy(out[0..r.s], content[0..r.s]); var write = r.s; for (0..count) |_| { @memcpy(out[write..][0..encoded_len], encoded[0..encoded_len]); write += encoded_len; } @memcpy(out[write..], content[r.e..]); return out; } // `~` / `` ` `` / ``Alt-` ``: case-map the inclusive range [a, b]. pub const CaseOp = enum { toggle, lower, upper }; pub fn changeCase(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, op: CaseOp) ![]u8 { const r = rangeBytes(content, a, b); const out = try alloc.dupe(u8, content); for (out[r.s..r.e]) |*p| { p.* = switch (op) { .toggle => if (std.ascii.isUpper(p.*)) std.ascii.toLower(p.*) else std.ascii.toUpper(p.*), .lower => std.ascii.toLower(p.*), .upper => std.ascii.toUpper(p.*), }; } return out; } // `J`: join line `row` with the next — the newline and the next line's leading // whitespace become one space (helix join). `col` is the space's column. // Null when `row` is the last line. pub fn joinLine(alloc: std.mem.Allocator, content: []const u8, row: usize) !?struct { content: []u8, col: usize } { if (row + 1 >= lineCount(content)) return null; const a = lineSlice(content, row); const next = lineSlice(content, row + 1); const b = std.mem.trimStart(u8, next, " \t"); const start = lineStartOffset(content, row); const rest = lineStartOffset(content, row + 1) + (next.len - b.len); const prefix_end = start + a.len; const out = try alloc.alloc(u8, prefix_end + 1 + content.len - rest); @memcpy(out[0..prefix_end], content[0..prefix_end]); out[prefix_end] = ' '; @memcpy(out[prefix_end + 1 ..], content[rest..]); return .{ .content = out, .col = a.len }; } // `>` / `<`: indent/unindent lines [r0, r1]. Fixed width — pardes has no // per-language indent config; 4 spaces, one tab counts as one level out. pub const INDENT_W = 4; pub fn indentLines(alloc: std.mem.Allocator, content: []const u8, r0: usize, r1: usize, add: bool) ![]u8 { const lo = @min(r0, r1); const hi = @max(r0, r1); var out_len = content.len; var it = std.mem.splitScalar(u8, content, '\n'); var row: usize = 0; while (it.next()) |line| : (row += 1) { if (row < lo or row > hi) continue; if (add) { if (line.len != 0) out_len += INDENT_W; } else { var cut: usize = 0; if (line.len > 0 and line[0] == '\t') { cut = 1; } else while (cut < line.len and cut < INDENT_W and line[cut] == ' ') cut += 1; out_len -= cut; } } const out = try alloc.alloc(u8, out_len); it = std.mem.splitScalar(u8, content, '\n'); row = 0; var write: usize = 0; while (it.next()) |line| : (row += 1) { if (row > 0) { out[write] = '\n'; write += 1; } var selected = line; if (row >= lo and row <= hi) { if (add) { if (line.len != 0) { @memset(out[write..][0..INDENT_W], ' '); write += INDENT_W; } } else if (line.len > 0 and line[0] == '\t') { selected = line[1..]; } else { var cut: usize = 0; while (cut < line.len and cut < INDENT_W and line[cut] == ' ') cut += 1; selected = line[cut..]; } } @memcpy(out[write..][0..selected.len], selected); write += selected.len; } return out; } // `Ctrl-a`/`Ctrl-x`: add `delta` to the decimal integer under the cursor // (helix: under the cursor only, no forward scan). Null when the cursor is // not on a number. The new cursor sits on the number's last digit. pub fn adjustNumber(alloc: std.mem.Allocator, content: []const u8, c: Cursor, delta: i64) !?struct { content: []u8, cur: Cursor } { const line = lineSlice(content, c.row); if (c.col >= line.len) return null; var s = c.col; var e = c.col; if (!std.ascii.isDigit(line[s])) { // sitting on the '-' of a negative number counts if (!(line[s] == '-' and s + 1 < line.len and std.ascii.isDigit(line[s + 1]))) return null; e = s + 1; } while (s > 0 and std.ascii.isDigit(line[s - 1])) s -= 1; if (s > 0 and line[s - 1] == '-') s -= 1; while (e < line.len and std.ascii.isDigit(line[e])) e += 1; const val = std.fmt.parseInt(i64, line[s..e], 10) catch return null; const nv = val +| delta; var buf: [24]u8 = undefined; // "{d}" prints '+' for positive signed ints — format the magnitude unsigned const numstr = if (nv < 0) std.fmt.bufPrint(&buf, "-{d}", .{@abs(nv)}) catch return null else std.fmt.bufPrint(&buf, "{d}", .{@abs(nv)}) catch return null; const off = lineStartOffset(content, c.row); const start = off + s; return .{ .content = try spliceAlloc(alloc, content, start, off + e, numstr), .cur = .{ .row = c.row, .col = s + numstr.len - 1 } }; } // delete the EXCLUSIVE span [a, b) — insert-mode kills. col may equal the // line length (the newline); a kill crossing it passes b = (row+1, 0). pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor) ![]u8 { const s = lineStartOffset(content, a.row) + @min(a.col, lineSlice(content, a.row).len); const e = lineStartOffset(content, b.row) + @min(b.col, lineSlice(content, b.row).len); if (e <= s) return alloc.dupe(u8, content); return spliceAlloc(alloc, content, s, e, ""); } // ---- helix range engine (phase 5) ---- // // Gap-offset ranges over the FLAT buffer, ported faithfully from // helix-core/src/movement.rs + selection.rs @ 278b24389 (the genizah // checkout). Positions are UTF-8 gap offsets 0..=text.len. Stored columns // remain byte offsets, but every range boundary is an extended-grapheme // boundary. A range with // head > anchor selects [anchor, head) with the block cursor ON head-1; // head < anchor selects [head, anchor) with the cursor ON head. The // differential suite (test/hxcases, `zig build hxdiff`) pins every behavior // here key-for-key against a real helix. pub const HxRange = struct { anchor: usize, head: usize }; /// The first byte of the grapheme cluster containing `off`. /// /// The general answer needs UAX #29, which is why the slow path below iterates from the start of /// `text` with the full break state machine - and that made this the single hottest function in a /// keystroke: 21.5% of a profiled edit at the ESP32-P4's 40x12 geometry, because the render path /// calls it once per visible row with a column offset, so the cost follows the cursor's distance /// along its line. That is exactly the shape measured on the die, where inserting at column 320 of /// a fixed line cost 7.8 ms more than inserting at column 0 of the same line. /// /// The fast path is sound rather than approximate. In UAX #29 every ASCII scalar is its own /// grapheme cluster with ONE exception, GB3: CR is joined to a following LF. Every other rule that /// could extend a cluster across `off` - Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator - /// is spelled with non-ASCII scalars. So if the byte at `off` and the byte before it are both /// ASCII and are not that CR-LF pair, `off` already IS a cluster boundary and there is nothing to /// search for. Text that is not all ASCII still takes the slow path, byte for byte as before. pub fn graphemeStart(text: []const u8, off: usize) usize { const bounded = @min(off, text.len); if (bounded == text.len) return text.len; if (text[bounded] < 0x80) { if (bounded == 0) return 0; const prev = text[bounded - 1]; if (prev < 0x80 and !(prev == '\r' and text[bounded] == '\n')) return bounded; } var it = uucode.grapheme.utf8Iterator(text); while (it.nextGrapheme()) |g| { if (bounded < g.end) return g.start; } return text.len; } /// one extended grapheme forward, clamped at text.len pub fn nextGrapheme(text: []const u8, off: usize) usize { if (off >= text.len) return text.len; // The editor's own offsets are already boundaries. Keep the overwhelmingly // common ASCII path O(1); only repair a continuation-byte input here. // // GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a // following LF into the same cluster. `graphemeStart` spells that exclusion // out (:875) and this did not, so the two disagreed about a CRLF file by // exactly one byte — a head stepped onto the offset between CR and LF and // `graphemeStart` then repaired it back onto the CR. Excluded here for the // same reason and in the same words; everything else ASCII is still O(1). if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80) and !(text[off] == '\r' and off + 1 < text.len and text[off + 1] == '\n')) return off + 1; var start = off; while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1; if (start != off) start = graphemeStart(text, off); var it = uucode.grapheme.utf8Iterator(text[start..]); const g = it.nextGrapheme() orelse return @min(start + 1, text.len); return start + g.end; } pub fn prevGrapheme(text: []const u8, off: usize) usize { var bounded = @min(off, text.len); if (bounded < text.len) { const repaired = graphemeStart(text, bounded); if (repaired < bounded) return repaired; bounded = repaired; } if (bounded == 0) return 0; // ...and the same GB3 exclusion, from the other side: a CR before this LF // means the cluster starts one byte earlier than the fast path would say. if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80) and !(bounded >= 2 and text[bounded - 2] == '\r' and text[bounded - 1] == '\n')) return bounded - 1; // Graphemes cannot cross a line break. Restrict the forward segmentation // needed for a reverse step to the current line instead of rescanning the // complete buffer. const line_start = if (std.mem.lastIndexOfScalar(u8, text[0 .. bounded - 1], '\n')) |nl| nl + 1 else 0; if (line_start == bounded) return bounded - 1; // the newline is its own editor cell var it = uucode.grapheme.utf8Iterator(text[line_start..bounded]); var last = line_start; while (it.nextGrapheme()) |g| last = line_start + g.start; return last; } /// Byte offset of the zero-based grapheme column, clamped to the line end. pub fn graphemeAtColumn(text: []const u8, column: usize) usize { var at: usize = 0; var col: usize = 0; while (at < text.len and col < column) : (col += 1) at = nextGrapheme(text, at); return at; } /// the block cursor cell of a range (helix Range::cursor) pub fn hxCursor(text: []const u8, r: HxRange) usize { return if (r.head > r.anchor) prevGrapheme(text, r.head) else r.head; } /// helix Range::put_cursor: park the block cursor at cell `idx`, optionally /// extending — the anchor shifts one grapheme when the head crosses it so the /// anchor CELL stays fixed. pub fn hxPutCursor(text: []const u8, r: HxRange, idx: usize, extend: bool) HxRange { if (!extend) return .{ .anchor = idx, .head = idx }; var anchor = r.anchor; if (r.head >= r.anchor and idx < r.anchor) { anchor = nextGrapheme(text, r.anchor); } else if (r.head < r.anchor and idx >= r.anchor) { anchor = prevGrapheme(text, r.anchor); } if (anchor <= idx) return .{ .anchor = anchor, .head = nextGrapheme(text, idx) }; return .{ .anchor = anchor, .head = idx }; } // ropey-style line math: len_lines = count('\n') + 1 — the slot after a // trailing '\n' is a real, empty last line and the cursor can sit there. pub fn hxLineCount(text: []const u8) usize { return std.mem.count(u8, text, "\n") + 1; } pub fn hxLineOf(text: []const u8, off: usize) usize { return std.mem.count(u8, text[0..@min(off, text.len)], "\n"); } /// offset of line's terminator ('\n'), or text.len on the last line pub fn hxLineEndIdx(text: []const u8, line: usize) usize { const s = lineStartOffset(text, line); return if (std.mem.indexOfScalarPos(u8, text, s, '\n')) |nl| nl else text.len; } /// gap offset -> (row, col) cell pub fn hxPos(text: []const u8, off: usize) Cursor { const bounded = @min(off, text.len); // The line start is the byte after the last '\n' BEFORE off, which is the // same number lineStartOffset(text, row) walks the whole prefix to reach — // one backward scan of a single line instead of a second pass over // everything above the cursor. On a multi-MB buffer that second pass was // most of what a keystroke cost. const s = if (std.mem.lastIndexOfScalar(u8, text[0..bounded], '\n')) |nl| nl + 1 else 0; const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; const col = graphemeStart(text[s..e], @min(bounded - s, e - s)); return .{ .row = hxLineOf(text, s + col), .col = col }; } /// (row, col) -> clamped gap offset; col == line length lands ON the '\n' pub fn hxOff(text: []const u8, c: Cursor) usize { const row = @min(c.row, hxLineCount(text) - 1); const s = lineStartOffset(text, row); // hxLineEndIdx(text, row) inlined: it starts by walking to `row` again, // and we are already standing there const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; const raw = @min(c.col, e - s); return s + graphemeStart(text[s..e], raw); } pub const WordTarget = enum { next_word_start, next_word_end, prev_word_start, prev_word_end, next_long_word_start, next_long_word_end, prev_long_word_start, prev_long_word_end, }; // helix categorize_char: Eol is its OWN category, distinct from Whitespace — // that distinction is load-bearing in reached_target. const HxCat = enum { word, punct, ws, eol }; fn hxCatAt(text: []const u8, off: usize) HxCat { if (off >= text.len) return .eol; const cp = codepointAt(text, off); if (cp == '\n' or cp == '\r') return .eol; return switch (kindOfCodepoint(cp)) { .word => .word, .punct => .punct, .ws => .ws, }; } fn hxIsWs(c: HxCat) bool { // Rust char::is_whitespace (includes line endings) return c == .ws or c == .eol; } fn hxIsWordBoundary(a: HxCat, b: HxCat) bool { return a != b; } fn hxIsLongBoundary(a: HxCat, b: HxCat) bool { if ((a == .word and b == .punct) or (a == .punct and b == .word)) return false; return a != b; } fn hxReached(target: WordTarget, prev: HxCat, next: HxCat) bool { return switch (target) { .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (next == .eol or !hxIsWs(next)), .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or next == .eol), .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (next == .eol or !hxIsWs(next)), .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or next == .eol), }; } fn wmIsPrev(t: WordTarget) bool { return switch (t) { .prev_word_start, .prev_word_end, .prev_long_word_start, .prev_long_word_end => true, else => false, }; } /// w/b/e/W/B/E: helix word_move — each step selects the traversed span. pub fn hxWordMove(text: []const u8, r0: HxRange, count: usize, target: WordTarget) HxRange { const is_prev = wmIsPrev(target); if ((is_prev and r0.head == 0) or (!is_prev and r0.head == text.len)) return r0; // block-cursor prep: collapse to the 1-wide cell at the head, pointing // in the motion direction (the anchor of the input is irrelevant) var r: HxRange = if (is_prev) (if (r0.anchor < r0.head) .{ .anchor = r0.head, .head = prevGrapheme(text, r0.head) } else .{ .anchor = nextGrapheme(text, r0.head), .head = r0.head }) else (if (r0.anchor < r0.head) .{ .anchor = prevGrapheme(text, r0.head), .head = r0.head } else .{ .anchor = r0.head, .head = nextGrapheme(text, r0.head) }); for (0..@max(1, count)) |_| { const next = hxRangeToTarget(text, target, r, is_prev); if (next.anchor == r.anchor and next.head == r.head) break; r = next; } return r; } // port of CharHelpers::range_to_target — a char iterator walking away from // origin.head; when reversed, "next" reads the byte just behind the position. fn hxRangeToTarget(text: []const u8, target: WordTarget, origin: HxRange, is_prev: bool) HxRange { var anchor = origin.anchor; var head = origin.head; var it = origin.head; var prev_cat: ?HxCat = if (is_prev) (if (it < text.len) hxCatAt(text, it) else null) else (if (it > 0) hxCatAt(text, prevGrapheme(text, it)) else null); // skip any initial newline characters while (true) { if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break; const cell = if (is_prev) prevGrapheme(text, it) else it; const cat = hxCatAt(text, cell); if (cat != .eol) break; it = if (is_prev) cell else nextGrapheme(text, cell); prev_cat = cat; head = it; } if (prev_cat == .eol) anchor = head; // find the target position const head_start = head; while (true) { if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break; const cell = if (is_prev) prevGrapheme(text, it) else it; const next_cat = hxCatAt(text, cell); if (prev_cat == null or hxReached(target, prev_cat.?, next_cat)) { if (head == head_start) anchor = head else break; } prev_cat = next_cat; it = if (is_prev) cell else nextGrapheme(text, cell); head = it; } return .{ .anchor = anchor, .head = head }; } /// a ropey "line is a line ending" — the line has no content of its own fn hxLineIsEmpty(text: []const u8, line: usize) bool { return lineStartOffset(text, line) == hxLineEndIdx(text, line); } /// ]p / [p: helix move_next_paragraph / move_prev_paragraph pub fn hxParaMove(text: []const u8, r: HxRange, count: usize, fwd: bool, extend: bool) HxRange { const nlines = hxLineCount(text); const cursor = hxCursor(text, r); var line = hxLineOf(text, cursor); if (fwd) { const nxt_start = if (line + 1 >= nlines) text.len else lineStartOffset(text, line + 1); const last_char = prevGrapheme(text, nxt_start) == cursor; const curr_empty = hxLineIsEmpty(text, line); const next_empty = hxLineIsEmpty(text, @min(nlines - 1, line + 1)); const curr_empty_to_line = curr_empty and !next_empty; // skip the character after the paragraph boundary if (curr_empty_to_line and last_char) line += 1; var l = line; var last_line = l; for (0..@max(1, count)) |_| { while (l < nlines and !hxLineIsEmpty(text, l)) l += 1; while (l < nlines and hxLineIsEmpty(text, l)) l += 1; if (l == last_line) break; last_line = l; } const head = if (l >= nlines) text.len else lineStartOffset(text, l); const anchor = if (extend) hxPutCursor(text, r, head, true).anchor else if (curr_empty_to_line and last_char) r.head else cursor; return .{ .anchor = anchor, .head = head }; } const first_char = lineStartOffset(text, line) == cursor; const prev_empty = hxLineIsEmpty(text, line -| 1); const curr_empty = hxLineIsEmpty(text, line); const prev_empty_to_line = prev_empty and !curr_empty; // skip the character before the paragraph boundary if (prev_empty_to_line and !first_char) line += 1; var l = line; var last_line = l; for (0..@max(1, count)) |_| { while (l > 0 and hxLineIsEmpty(text, l - 1)) l -= 1; while (l > 0 and !hxLineIsEmpty(text, l - 1)) l -= 1; if (l == last_line) break; last_line = l; } const head = lineStartOffset(text, l); const anchor = if (extend) hxPutCursor(text, r, head, true).anchor else if (prev_empty_to_line and first_char) cursor else r.head; return .{ .anchor = anchor, .head = head }; } /// j/k target: helix move_vertically — goal_col clamps to the line's content /// length, i.e. the cursor may land ON the '\n' of a shorter line. pub fn hxVertTarget(text: []const u8, pos: usize, down: bool, count: usize, goal_col: usize) usize { const nlines = hxLineCount(text); const line = hxLineOf(text, pos); const nline = if (down) @min(line + @max(1, count), nlines - 1) else line -| @max(1, count); const s = lineStartOffset(text, nline); // hxLineEndIdx(text, nline) without its second walk to nline (see hxOff) const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; return s + graphemeStart(text[s..e], @min(goal_col, e - s)); } /// f/F/t/T target cell. helix find_char: the exclusive (till) search starts /// one further out so repeats make progress; not-found = null (no move). pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u21, fwd: bool, till: bool, count: usize) ?usize { var left = @max(1, count); if (fwd) { const head = nextGrapheme(text, cursor); var i = if (till) nextGrapheme(text, head) else head; if (i > text.len) return null; while (i < text.len) : (i = nextGrapheme(text, i)) { if (codepointAt(text, i) == ch) { left -= 1; if (left == 0) return if (till) prevGrapheme(text, i) else i; } } return null; } var i = if (till) prevGrapheme(text, cursor) else cursor; while (i > 0) { i = prevGrapheme(text, i); if (codepointAt(text, i) == ch) { left -= 1; if (left == 0) return if (till) nextGrapheme(text, i) else i; } } return null; } // helix textobject.rs find_word_boundary fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usize { var prev: HxCat = if (fwd) (if (pos0 == 0) .ws else hxCatAt(text, prevGrapheme(text, pos0))) else (if (pos0 >= text.len) .ws else hxCatAt(text, pos0)); var pos = pos0; var it = pos0; while (true) { if ((fwd and it >= text.len) or (!fwd and it == 0)) break; const cell = if (fwd) it else prevGrapheme(text, it); const cat = hxCatAt(text, cell); if (cat == .eol or cat == .ws) return pos; if (!long and cat != prev and pos != 0 and pos != text.len) return pos; it = if (fwd) nextGrapheme(text, cell) else cell; pos = it; prev = cat; } return pos; } /// miw/maw (and W): helix textobject_word — on whitespace it selects the /// whitespace run's boundary (a 1-wide cursor there) pub fn hxTextobjectWord(text: []const u8, r: HxRange, around: bool, long: bool) HxRange { const pos = hxCursor(text, r); const word_start = hxFindWordBoundary(text, pos, false, long); const cat: HxCat = if (pos < text.len) hxCatAt(text, pos) else .ws; const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, nextGrapheme(text, pos), true, long); if (word_start == word_end or !around) return .{ .anchor = word_start, .head = word_end }; var end = word_end; while (end < text.len and hxIsWs(hxCatAt(text, end)) and hxCatAt(text, end) != .eol) end = nextGrapheme(text, end); if (end > word_end) return .{ .anchor = word_start, .head = end }; var start = word_start; while (start > 0) { const before = prevGrapheme(text, start); const before_cat = hxCatAt(text, before); if (!hxIsWs(before_cat) or before_cat == .eol) break; start = before; } return .{ .anchor = start, .head = word_end }; } /// mip/map: helix textobject_paragraph pub fn hxTextobjectParagraph(text: []const u8, r: HxRange, around: bool, count: usize) HxRange { const nlines = hxLineCount(text); const cursor = hxCursor(text, r); var line = hxLineOf(text, cursor); const prev_empty = hxLineIsEmpty(text, line -| 1); const curr_empty = hxLineIsEmpty(text, line); const next_empty = line + 1 >= nlines or hxLineIsEmpty(text, line + 1); const nxt_start = if (line + 1 >= nlines) text.len else lineStartOffset(text, line + 1); const last_char = prevGrapheme(text, nxt_start) == cursor; const prev_empty_to_line = prev_empty and !curr_empty; const curr_empty_to_line = curr_empty and !next_empty; var line_back = line; if (prev_empty_to_line or curr_empty_to_line) line_back += 1; // do not include the current paragraph on a paragraph end (include next) if (!(curr_empty_to_line and last_char)) { while (line_back > 0 and hxLineIsEmpty(text, line_back - 1)) line_back -= 1; while (line_back > 0 and !hxLineIsEmpty(text, line_back - 1)) line_back -= 1; } if (curr_empty_to_line and last_char) line += 1; const n = @max(1, count); var count_done: usize = 0; for (0..n) |_| { var done = false; while (line < nlines and !hxLineIsEmpty(text, line)) { line += 1; done = true; } while (line < nlines and hxLineIsEmpty(text, line)) line += 1; if (done) count_done += 1; } // search one paragraph backwards when we ran off the end if (count_done != n and line >= nlines) { while (line_back > 0 and hxLineIsEmpty(text, line_back - 1)) line_back -= 1; while (line_back > 0 and !hxLineIsEmpty(text, line_back - 1)) line_back -= 1; } if (!around) { // inside: drop the trailing whitespace paragraph while (line > 0 and hxLineIsEmpty(text, line - 1)) line -= 1; } return .{ .anchor = lineStartOffset(text, line_back), .head = if (line >= nlines) text.len else lineStartOffset(text, line), }; } test "hx textobject word and paragraph" { const t = "alpha beta gamma\n"; // miw mid-word var r = hxTextobjectWord(t, .{ .anchor = 8, .head = 9 }, false, false); try std.testing.expectEqual(@as(usize, 6), r.anchor); try std.testing.expectEqual(@as(usize, 10), r.head); // maw on the space after "beta": collapses to the boundary r = hxTextobjectWord(t, .{ .anchor = 10, .head = 11 }, true, false); try std.testing.expectEqual(@as(usize, 10), r.anchor); try std.testing.expectEqual(@as(usize, 10), r.head); const t2 = "aa\n\ncc\n"; // mip from the blank line selects the NEXT paragraph r = hxTextobjectParagraph(t2, .{ .anchor = 3, .head = 4 }, false, 1); try std.testing.expectEqual(@as(usize, 4), r.anchor); try std.testing.expectEqual(@as(usize, 7), r.head); } /// leading-whitespace visual width (tab -> next multiple of INDENT_W) pub fn hxIndentWidth(line: []const u8) usize { var w: usize = 0; for (line) |ch| { if (ch == ' ') w += 1 else if (ch == '\t') w = (w / INDENT_W + 1) * INDENT_W else break; } return w; } /// full indent LEVELS of a line as spaces (helix indent_level_for_line: /// partial levels round down) — what o/O/insert-newline copy. pub fn hxIndentString(line: []const u8) []const u8 { const level = hxIndentWidth(line) / INDENT_W; const max = " "; // 8 levels is plenty (ponytail) return max[0..@min(level * INDENT_W, max.len)]; } /// Indent width for an inserted newline. Keep the current full indent levels, /// then add one logical tab after a simple delimiter-shaped line ending. /// `)` intentionally includes both ordinary calls and the requested `})` /// continuation shape; this is syntax-agnostic and does not try to parse. pub fn hxNewlineIndentWidth(line: []const u8, col: usize) usize { const prefix = std.mem.trimEnd(u8, line[0..@min(col, line.len)], " \t"); const extra = if (prefix.len == 0) false else switch (prefix[prefix.len - 1]) { '(', '[', '{', ')' => true, else => false, }; return hxIndentString(line).len + @as(usize, if (extra) INDENT_W else 0); } test "newline indent keeps levels and adds one after delimiters" { try std.testing.expectEqual(@as(usize, 4), hxNewlineIndentWidth(" value", 9)); try std.testing.expectEqual(@as(usize, 8), hxNewlineIndentWidth(" call()", 10)); try std.testing.expectEqual(@as(usize, 8), hxNewlineIndentWidth(" callback({}) ", 16)); try std.testing.expectEqual(@as(usize, 4), hxNewlineIndentWidth("work(", 5)); try std.testing.expectEqual(@as(usize, 4), hxNewlineIndentWidth("list[tail", 5)); } /// helix Ctrl-a / Ctrl-x: increment the SELECTED text as a decimal integer. /// Zero-padding is preserved (width follows sign flips, helix-style). /// Ponytail: no 0x/0o/0b bases, no '_' separators — decimal only. pub fn hxIncrement(alloc: std.mem.Allocator, frag: []const u8, amount: i64) !?[]u8 { if (frag.len == 0) return null; const neg = frag[0] == '-'; const digits = if (neg) frag[1..] else frag; if (digits.len == 0) return null; for (digits) |ch| if (!std.ascii.isDigit(ch)) return null; const val = std.fmt.parseInt(i128, frag, 10) catch return null; const nv = val +| @as(i128, amount); const pad = digits[0] == '0'; const neg_after = nv < 0; // format_length includes the sign, adjusted when the sign flips var flen: usize = frag.len; if (neg and !neg_after) flen -= 1; if (!neg and neg_after) flen += 1; var buf: [48]u8 = undefined; // "{d}" prints '+' for positive signed ints — format the magnitude unsigned const mag = std.fmt.bufPrint(&buf, "{d}", .{@abs(nv)}) catch return null; const sign_len: usize = @intFromBool(neg_after); const want = if (pad) flen - sign_len else mag.len; const digits_len = @max(want, mag.len); const out = try alloc.alloc(u8, sign_len + digits_len); var write: usize = 0; if (neg_after) { out[0] = '-'; write = 1; } @memset(out[write..][0 .. digits_len - mag.len], '0'); write += digits_len - mag.len; @memcpy(out[write..], mag); return out; } test "hx word moves match helix" { const t = "alpha beta\n"; // w from a fresh 1-wide cursor selects "alpha " (cursor on the space) var r = hxWordMove(t, .{ .anchor = 0, .head = 1 }, 1, .next_word_start); try std.testing.expectEqual(@as(usize, 0), r.anchor); try std.testing.expectEqual(@as(usize, 6), r.head); // e from the same start ends on 'a' of alpha r = hxWordMove(t, .{ .anchor = 0, .head = 1 }, 1, .next_word_end); try std.testing.expectEqual(@as(usize, 5), r.head); try std.testing.expectEqual(@as(usize, 0), r.anchor); // b from the w result selects "alpha" backward r = hxWordMove(t, .{ .anchor = 6, .head = 10 }, 1, .prev_word_start); try std.testing.expectEqual(@as(usize, 10), r.anchor); try std.testing.expectEqual(@as(usize, 6), r.head); // 2w on "one two three": anchor comes from the last hop only const t2 = "one two three\n"; r = hxWordMove(t2, .{ .anchor = 0, .head = 1 }, 2, .next_word_start); try std.testing.expectEqual(@as(usize, 4), r.anchor); try std.testing.expectEqual(@as(usize, 8), r.head); // w at EOF collapses to a zero-width range at len const t3 = "alpha\n"; r = hxWordMove(t3, .{ .anchor = 0, .head = 5 }, 1, .next_word_start); try std.testing.expectEqual(@as(usize, 6), r.head); try std.testing.expectEqual(@as(usize, 6), r.anchor); // W treats punct runs as word chars const t4 = "foo.bar baz\n"; r = hxWordMove(t4, .{ .anchor = 0, .head = 1 }, 1, .next_long_word_start); try std.testing.expectEqual(@as(usize, 0), r.anchor); try std.testing.expectEqual(@as(usize, 8), r.head); } test "hx put cursor keeps the anchor cell across crossings" { const t = "abcdef\n"; // forward range [2,3) extended left of the anchor: anchor cell stays 2 var r = hxPutCursor(t, .{ .anchor = 2, .head = 3 }, 0, true); try std.testing.expectEqual(@as(usize, 3), r.anchor); try std.testing.expectEqual(@as(usize, 0), r.head); try std.testing.expectEqual(@as(usize, 0), hxCursor(t, r)); // and back: cursor to 4 -> forward again, anchor gap back to 2 r = hxPutCursor(t, r, 4, true); try std.testing.expectEqual(@as(usize, 2), r.anchor); try std.testing.expectEqual(@as(usize, 5), r.head); } test "hx paragraph moves" { const t = "aa\nbb\n\ncc\ndd\n\nee\n"; // ]p from the top selects through the blank line to the next block var r = hxParaMove(t, .{ .anchor = 0, .head = 1 }, 1, true, false); try std.testing.expectEqual(@as(usize, 0), r.anchor); try std.testing.expectEqual(@as(usize, 7), r.head); // [p from "ee" (line 6, offset 14) goes back to "cc" block start r = hxParaMove(t, .{ .anchor = 14, .head = 15 }, 1, false, false); try std.testing.expectEqual(@as(usize, 14), r.anchor); try std.testing.expectEqual(@as(usize, 7), r.head); } test "hx vertical: goal col clamps onto the newline cell" { const t = "abcdef\nab\nabcdef\n"; // from (0,5) down: line "ab" clamps to its '\n' at offset 9 try std.testing.expectEqual(@as(usize, 9), hxVertTarget(t, 5, true, 1, 5)); // two down with the same goal restores col 5 try std.testing.expectEqual(@as(usize, 15), hxVertTarget(t, 9, true, 1, 5)); } test "hx find targets" { const t = "abcabc\n"; try std.testing.expectEqual(@as(usize, 2), hxFindTarget(t, 0, 'c', true, false, 1).?); try std.testing.expectEqual(@as(usize, 5), hxFindTarget(t, 0, 'c', true, false, 2).?); try std.testing.expectEqual(@as(usize, 1), hxFindTarget(t, 0, 'c', true, true, 1).?); // till repeat skips the adjacent target: from cell 1, next tc reaches 4 try std.testing.expectEqual(@as(usize, 4), hxFindTarget(t, 1, 'c', true, true, 1).?); try std.testing.expectEqual(@as(usize, 3), hxFindTarget(t, 5, 'a', false, false, 1).?); try std.testing.expectEqual(@as(usize, 4), hxFindTarget(t, 5, 'a', false, true, 1).?); try std.testing.expectEqual(@as(?usize, null), hxFindTarget(t, 0, 'z', true, false, 1)); } test "extended grapheme boundaries cover combining emoji flag and CJK text" { const text = "a" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "🇧🇷" ++ "界"; const boundaries = [_]usize{ 0, 1, 4, 19, 27, 30 }; for (boundaries[0 .. boundaries.len - 1], boundaries[1..]) |start, end| { try std.testing.expectEqual(end, nextGrapheme(text, start)); try std.testing.expectEqual(start, prevGrapheme(text, end)); } // Stale byte offsets are repaired to a cluster boundary instead of being // allowed to leak continuation bytes into cursor state. try std.testing.expectEqual(@as(usize, 1), graphemeStart(text, 2)); try std.testing.expectEqual(@as(usize, 1), prevGrapheme(text, 3)); try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3)); } test "the ASCII arms of graphemeStart and nextGrapheme agree with the UAX #29 walk" { // Both functions answer ASCII from arithmetic and hand everything else to the segmenter. The // guard is a claim about UAX #29 (an ASCII scalar is its own cluster unless the next scalar // extends it, and every extender is non-ASCII), so pin it against the walk it skips rather // than against transcribed offsets: same text, both routes, every offset including past the end. const H = struct { // `graphemeStart` with the ASCII arm deleted — nothing else changed. fn start(text: []const u8, off: usize) usize { const bounded = @min(off, text.len); if (bounded == text.len) return text.len; var it = uucode.grapheme.utf8Iterator(text); while (it.nextGrapheme()) |g| { if (bounded < g.end) return g.start; } return text.len; } // `nextGrapheme` with the ASCII arm deleted. fn next(text: []const u8, off: usize) usize { if (off >= text.len) return text.len; var s = off; while (s > 0 and (text[s] & 0xC0) == 0x80) s -= 1; if (s != off) s = start(text, off); var it = uucode.grapheme.utf8Iterator(text[s..]); const g = it.nextGrapheme() orelse return @min(s + 1, text.len); return s + g.end; } fn check(text: []const u8) !void { var off: usize = 0; while (off <= text.len + 2) : (off += 1) { std.testing.expectEqual(start(text, off), graphemeStart(text, off)) catch |e| { std.debug.print("graphemeStart({any}, {d})\n", .{ text, off }); return e; }; std.testing.expectEqual(next(text, off), nextGrapheme(text, off)) catch |e| { std.debug.print("nextGrapheme({any}, {d})\n", .{ text, off }); return e; }; } } }; // Scalars that extend a preceding ASCII base into ONE cluster, which is the whole reason the // fast path inspects its neighbour: a combining mark, a ZWJ sequence, a spacing mark // (Devanagari visarga), a variation selector. Plus wide glyphs, a regional-indicator pair, // and three shapes of invalid UTF-8 the segmenter must still be trusted with: a bad start // byte, a truncated tail, a bad continuation. const neighbours = [_][]const u8{ "", "a", "\u{301}", "\u{200d}\u{1f680}", "\u{903}", "\u{fe0f}", "\u{20e3}", "\u{4e16}\u{754c}", "\u{1f642}", "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8", "\xe4\x28\xb8", }; // Every byte the range test can see, ASCII and not: 0x20..0x7e take the fast path, and \t, \r, // the rest of the C0 controls and DEL are excluded by it and must still reach the same answer. var buf: [16]u8 = undefined; var b: u8 = 0; while (b < 0x80) : (b += 1) { buf[0] = b; for (neighbours) |tail| { @memcpy(buf[1..][0..tail.len], tail); try H.check(buf[0 .. 1 + tail.len]); // ...and the same byte as a follower, so a boundary is probed from both sides. @memcpy(buf[0..tail.len], tail); buf[tail.len] = b; try H.check(buf[0 .. tail.len + 1]); } } // Text that has no CR-LF pair in it: GB3 is the one ASCII-only rule that joins two clusters, // and it gets its own test below because it is the single exclusion every fast path has to // carry by hand. for ([_][]const u8{ "a\r", "\ra", "\n\r", "a\rb\nc" }) |text| try H.check(text); // Mixed text long enough that a fast-path run starts, ends and restarts inside one string. try H.check("plain ascii then \u{4e16}\u{754c} then e\u{301} then more ascii"); } // GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a following LF into the same // cluster. Each of the three steppers carries that exclusion separately - `graphemeStart` at :875, // `nextGrapheme`'s ASCII arm at :896, `prevGrapheme`'s at :916 - so nothing but a test keeps them // agreeing. The invariant is that all three answer the same CRLF boundary: for every cluster the // segmenter reports, `graphemeStart` maps its start to itself, `nextGrapheme` maps that start to // its end, and `prevGrapheme` maps its end back to the start. // // This was a live bug: `nextGrapheme` and `prevGrapheme` stepped exactly one byte whenever the // byte at the offset and its neighbour were ASCII, so on a CRLF file the flat-buffer range engine // could step a head to offset 1 and `graphemeStart` would repair that same offset back to 0. Both // arms now spell the exclusion out, and this test is what holds them there. test "GB3 keeps CR-LF one cluster for every grapheme step" { const text = "a\r\nb"; // The reference: the same segmentation the slow arms of these functions run. var it = uucode.grapheme.utf8Iterator(text); var starts: [8]usize = undefined; var ends: [8]usize = undefined; var n: usize = 0; while (it.nextGrapheme()) |g| : (n += 1) { starts[n] = g.start; ends[n] = g.end; } for (starts[0..n], ends[0..n]) |start, end| { try std.testing.expectEqual(start, graphemeStart(text, start)); try std.testing.expectEqual(end, nextGrapheme(text, start)); try std.testing.expectEqual(start, prevGrapheme(text, end)); } } test "Unicode find and word motion stay on grapheme boundaries" { const lines = [_][]const u8{"\u{e9}x\u{e9}"}; try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'é', true, false, 1).?); const text = "café 世界 ok\n"; const first = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 1, .next_word_start); try std.testing.expectEqual(@as(usize, 0), first.anchor); try std.testing.expectEqual(@as(usize, 6), first.head); const second = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 2, .next_word_start); try std.testing.expectEqual(@as(usize, 6), second.anchor); try std.testing.expectEqual(@as(usize, 13), second.head); // Long-word motions split on Unicode whitespace, not only ASCII spaces. const nbsp = "alpha\u{a0}beta\n"; const long = hxWordMove(nbsp, .{ .anchor = 0, .head = 1 }, 1, .next_long_word_start); try std.testing.expectEqual(@as(usize, 7), long.head); } test "hx increment" { const a = std.testing.allocator; { const r = (try hxIncrement(a, "15", 1)).?; defer a.free(r); try std.testing.expectEqualStrings("16", r); } { const r = (try hxIncrement(a, "007", 1)).?; defer a.free(r); try std.testing.expectEqualStrings("008", r); } { const r = (try hxIncrement(a, "-3", 1)).?; defer a.free(r); try std.testing.expectEqualStrings("-2", r); } { const r = (try hxIncrement(a, "9", -10)).?; defer a.free(r); try std.testing.expectEqualStrings("-1", r); } try std.testing.expectEqual(@as(?[]u8, null), try hxIncrement(a, "a 1", 1)); try std.testing.expectEqual(@as(?[]u8, null), try hxIncrement(a, "", 1)); } // ---- tests ---- test "kindOf" { try std.testing.expectEqual(Kind.word, kindOf('a')); try std.testing.expectEqual(Kind.word, kindOf('_')); try std.testing.expectEqual(Kind.word, kindOf('9')); try std.testing.expectEqual(Kind.punct, kindOf('.')); try std.testing.expectEqual(Kind.punct, kindOf('(')); try std.testing.expectEqual(Kind.ws, kindOf(' ')); try std.testing.expectEqual(Kind.ws, kindOf('\n')); } test "char/line motions" { const lines = [_][]const u8{ "alpha beta", " two words", "x" }; const c = Cursor{ .row = 0, .col = 5 }; try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(&.{"hello"}, c)); try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, charRight(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 1, .col = 5 }, lineDown(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, lineUp(&lines, Cursor{ .row = 1, .col = 5 })); // line ends try std.testing.expectEqual(Cursor{ .row = 0, .col = 9 }, lineEnd(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 2, .col = 0 }, lineEnd(&lines, Cursor{ .row = 2, .col = 0 })); // first non-ws try std.testing.expectEqual(Cursor{ .row = 1, .col = 2 }, firstNonWsOf(&lines, Cursor{ .row = 1, .col = 0 })); // cursor row past the content (mouse click below a short pane): no panic try std.testing.expectEqual(Cursor{ .row = 24, .col = 0 }, firstNonWsOf(&lines, Cursor{ .row = 24, .col = 3 })); try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, paragraphBwd(&lines, Cursor{ .row = 24, .col = 0 })); try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, paragraphBwd(&[_][]const u8{}, Cursor{ .row = 5, .col = 0 })); } test "word motions w/b/e" { const lines = [_][]const u8{"this is a test"}; const w = &lines; // "this is a test", indices 0..13 try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, nextWordStart(w, Cursor{ .row = 0, .col = 0 }, false)); // t->next word "is" try std.testing.expectEqual(Cursor{ .row = 0, .col = 8 }, nextWordStart(w, Cursor{ .row = 0, .col = 5 }, false)); // -> "a" try std.testing.expectEqual(Cursor{ .row = 0, .col = 10 }, nextWordStart(w, Cursor{ .row = 0, .col = 8 }, false)); // -> "test" try std.testing.expectEqual(Cursor{ .row = 0, .col = 10 }, nextWordStart(w, Cursor{ .row = 0, .col = 9 }, false)); // from ws // b try std.testing.expectEqual(Cursor{ .row = 0, .col = 8 }, prevWordStart(w, Cursor{ .row = 0, .col = 10 }, false)); // test -> "a" try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, prevWordStart(w, Cursor{ .row = 0, .col = 8 }, false)); // -> "is" try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, prevWordStart(w, Cursor{ .row = 0, .col = 5 }, false)); // -> "this" // e try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, nextWordEnd(w, Cursor{ .row = 0, .col = 0 }, false)); // this[3] try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, nextWordEnd(w, Cursor{ .row = 0, .col = 3 }, false)); // -> "is"[6] try std.testing.expectEqual(Cursor{ .row = 0, .col = 13 }, nextWordEnd(w, Cursor{ .row = 0, .col = 10 }, false)); // -> "test"[13] } test "word motions cross line" { const lines = [_][]const u8{ "foo bar", "", "baz" }; const w = &lines; // from end of "foo bar" (row0 col6) w crosses the blank line to "baz" try std.testing.expectEqual(Cursor{ .row = 2, .col = 0 }, nextWordStart(w, Cursor{ .row = 0, .col = 6 }, false)); // b from "baz" crosses back to "bar" try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, prevWordStart(w, Cursor{ .row = 2, .col = 0 }, false)); // e from row0 col0 -> "foo" end (col2) try std.testing.expectEqual(Cursor{ .row = 0, .col = 2 }, nextWordEnd(w, Cursor{ .row = 0, .col = 0 }, false)); } test "lineSpan agrees with the whole-document scans it replaces" { // The bounded scan is only worth having if it is indistinguishable from the pair it replaced, // including at the edges that make line counting awkward: an empty document, a trailing // newline (which is its own empty last line), and a row past the end. for ([_][]const u8{ "", "a", "a\n", "a\nbb\n", "a\nbb\nccc", "\n", "\n\n" }) |content| { const n = lineCount(content); var row: usize = 0; while (row < n) : (row += 1) { const span = lineSpan(content, row) orelse { std.debug.print("row {d} of {s} missing\n", .{ row, content }); return error.MissingRow; }; try std.testing.expectEqual(lineStartOffset(content, row), span.start); try std.testing.expectEqualStrings(lineSlice(content, row), content[span.start..span.end]); } // One past the last line must be absent, which is what lets insertAt clamp. try std.testing.expectEqual(@as(?LineSpan, null), lineSpan(content, n)); } } test "insertAt still clamps a row past the end onto the last line" { const gpa = std.testing.allocator; const content = "a\nbb\nccc"; const out = try insertAt(gpa, content, .{ .row = 99, .col = 99 }, "X"); defer gpa.free(out); try std.testing.expectEqualStrings("a\nbb\ncccX", out); // And an in-range insert lands where the old spelling put it. const mid = try insertAt(gpa, content, .{ .row = 1, .col = 1 }, "X"); defer gpa.free(mid); try std.testing.expectEqualStrings("a\nbXb\nccc", mid); } test "long word W treats punct as word" { // "foo.bar baz" : W from 0 -> "baz" at 8 (foo.bar is one long word) const lines = [_][]const u8{"foo.bar baz"}; const w = &lines; try std.testing.expectEqual(Cursor{ .row = 0, .col = 8 }, nextWordStart(w, Cursor{ .row = 0, .col = 0 }, true)); // w (non-long) from 0 -> '.' at 3 (punct is its own word, like vim/helix) try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, nextWordStart(w, Cursor{ .row = 0, .col = 0 }, false)); } test "goto" { const lines = [_][]const u8{ "a", "b", "c" }; try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, gotoFirst()); try std.testing.expectEqual(Cursor{ .row = 2, .col = 0 }, gotoLast(&lines)); } test "lineStartOffset + lineSlice" { const content = "alpha\nbeta\n\ngamma"; try std.testing.expectEqual(@as(usize, 0), lineStartOffset(content, 0)); try std.testing.expectEqual(@as(usize, 6), lineStartOffset(content, 1)); try std.testing.expectEqual(@as(usize, 11), lineStartOffset(content, 2)); try std.testing.expectEqual(@as(usize, 12), lineStartOffset(content, 3)); try std.testing.expectEqual(@as(usize, 17), lineStartOffset(content, 4)); // past end try std.testing.expectEqualStrings("alpha", lineSlice(content, 0)); try std.testing.expectEqualStrings("beta", lineSlice(content, 1)); try std.testing.expectEqualStrings("", lineSlice(content, 2)); try std.testing.expectEqualStrings("gamma", lineSlice(content, 3)); try std.testing.expectEqual(@as(usize, 4), lineCount(content)); } test "insertAt mid-line and at end" { const content = "hello world"; const a = std.testing.allocator; const r1 = try insertAt(a, content, .{ .row = 0, .col = 5 }, "!"); defer a.free(r1); try std.testing.expectEqualStrings("hello! world", r1); const r2 = try insertAt(a, content, .{ .row = 0, .col = 99 }, "!"); defer a.free(r2); try std.testing.expectEqualStrings("hello world!", r2); } test "insertAt multiline creates lines" { const content = "a\nb"; const a = std.testing.allocator; const r = try insertAt(a, content, .{ .row = 0, .col = 1 }, "X\nY"); defer a.free(r); try std.testing.expectEqualStrings("aX\nY\nb", r); try std.testing.expectEqual(@as(usize, 3), lineCount(r)); } test "deleteChar" { const content = "abc"; const a = std.testing.allocator; const r = try deleteChar(a, content, .{ .row = 0, .col = 1 }); defer a.free(r); try std.testing.expectEqualStrings("ac", r); // past end: no-op const r2 = try deleteChar(a, content, .{ .row = 0, .col = 5 }); defer a.free(r2); try std.testing.expectEqualStrings("abc", r2); } test "Unicode edits replace and delete whole graphemes" { const a = std.testing.allocator; const content = "A" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "🇧🇷" ++ "界" ++ "Z"; const deleted = try deleteChar(a, content, .{ .row = 0, .col = 2 }); defer a.free(deleted); try std.testing.expectEqualStrings("A👩🏽\u{200d}🚀🇧🇷界Z", deleted); const replaced = try replaceChars(a, content, .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 19 }, '界'); defer a.free(replaced); try std.testing.expectEqualStrings("A界界界界Z", replaced); } test "deleteLines middle" { const content = "one\ntwo\nthree\nfour"; const a = std.testing.allocator; const d = try deleteLines(a, content, 1, 2); defer a.free(d.content); defer a.free(d.deleted); try std.testing.expectEqualStrings("one\nfour", d.content); try std.testing.expectEqualStrings("two\nthree", d.deleted); } test "deleteLines last line drops preceding newline" { const content = "one\ntwo\nthree"; const a = std.testing.allocator; const d = try deleteLines(a, content, 2, 2); defer a.free(d.content); defer a.free(d.deleted); try std.testing.expectEqualStrings("one\ntwo", d.content); try std.testing.expectEqualStrings("three", d.deleted); } test "deleteLines only line" { const content = "only"; const a = std.testing.allocator; const d = try deleteLines(a, content, 0, 0); defer a.free(d.content); defer a.free(d.deleted); try std.testing.expectEqualStrings("", d.content); try std.testing.expectEqualStrings("only", d.deleted); } test "rangeText + deleteRange (char-wise select)" { const a = std.testing.allocator; const content = "hello\nworld\nfoo"; // same-line inclusive range: "hello"[1..3] -> "ell" const t1 = try rangeText(a, content, .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 3 }); defer a.free(t1); try std.testing.expectEqualStrings("ell", t1); // reversed cursors give the same range const t2 = try rangeText(a, content, .{ .row = 0, .col = 3 }, .{ .row = 0, .col = 1 }); defer a.free(t2); try std.testing.expectEqualStrings("ell", t2); // cross-line range includes the newline: row0 col3 .. row1 col1 -> "lo\nwo" const t3 = try rangeText(a, content, .{ .row = 0, .col = 3 }, .{ .row = 1, .col = 1 }); defer a.free(t3); try std.testing.expectEqualStrings("lo\nwo", t3); // delete the same cross-line range const d = try deleteRange(a, content, .{ .row = 0, .col = 3 }, .{ .row = 1, .col = 1 }); defer a.free(d.content); defer a.free(d.deleted); try std.testing.expectEqualStrings("helrld\nfoo", d.content); try std.testing.expectEqualStrings("lo\nwo", d.deleted); } test "clearLine" { const content = "keep\nzap me\nkeep2"; const a = std.testing.allocator; const r = try clearLine(a, content, 1); defer a.free(r); try std.testing.expectEqualStrings("keep\n\nkeep2", r); } test "findChar f/F/t/T across lines and counts" { const lines = [_][]const u8{ "abcabc", "xa" }; const w = &lines; // f: next occurrence, on it try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(w, .{ .row = 0, .col = 0 }, 'a', true, false, 1).?); // count: 2fa crosses into the next line try std.testing.expectEqual(Cursor{ .row = 1, .col = 1 }, findChar(w, .{ .row = 0, .col = 0 }, 'a', true, false, 2).?); // t stops one short try std.testing.expectEqual(Cursor{ .row = 0, .col = 2 }, findChar(w, .{ .row = 0, .col = 0 }, 'a', true, true, 1).?); // F backward, on it try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, findChar(w, .{ .row = 0, .col = 3 }, 'a', false, false, 1).?); // T backward stops one after try std.testing.expectEqual(Cursor{ .row = 0, .col = 1 }, findChar(w, .{ .row = 0, .col = 3 }, 'a', false, true, 1).?); // not found: null, no move try std.testing.expectEqual(@as(?Cursor, null), findChar(w, .{ .row = 0, .col = 0 }, 'z', true, false, 1)); } test "matchBracket nesting both directions" { const lines = [_][]const u8{"a (b (c) d) e"}; const w = &lines; try std.testing.expectEqual(Cursor{ .row = 0, .col = 10 }, matchBracket(w, .{ .row = 0, .col = 2 }).?); try std.testing.expectEqual(Cursor{ .row = 0, .col = 2 }, matchBracket(w, .{ .row = 0, .col = 10 }).?); try std.testing.expectEqual(Cursor{ .row = 0, .col = 7 }, matchBracket(w, .{ .row = 0, .col = 5 }).?); try std.testing.expectEqual(@as(?Cursor, null), matchBracket(w, .{ .row = 0, .col = 0 })); } test "matchBracket across lines" { const lines = [_][]const u8{ "if (x) {", " y", "}" }; const w = &lines; try std.testing.expectEqual(Cursor{ .row = 2, .col = 0 }, matchBracket(w, .{ .row = 0, .col = 7 }).?); try std.testing.expectEqual(Cursor{ .row = 0, .col = 7 }, matchBracket(w, .{ .row = 2, .col = 0 }).?); } test "paragraph motions" { const lines = [_][]const u8{ "one", "two", "", "", "three", "four", "", "five" }; const w = &lines; try std.testing.expectEqual(Cursor{ .row = 4, .col = 0 }, paragraphFwd(w, .{ .row = 0, .col = 1 })); try std.testing.expectEqual(Cursor{ .row = 7, .col = 0 }, paragraphFwd(w, .{ .row = 4, .col = 0 })); // no next block: the last line try std.testing.expectEqual(Cursor{ .row = 7, .col = 0 }, paragraphFwd(w, .{ .row = 7, .col = 0 })); // from mid-block up to its start; from a start up to the previous block's try std.testing.expectEqual(Cursor{ .row = 4, .col = 0 }, paragraphBwd(w, .{ .row = 5, .col = 1 })); try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, paragraphBwd(w, .{ .row = 4, .col = 0 })); try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, paragraphBwd(w, .{ .row = 0, .col = 0 })); } test "pairRange inside/around, cursor on and between brackets" { const lines = [_][]const u8{"f(a, (b))"}; const w = &lines; const around = pairRange(w, .{ .row = 0, .col = 3 }, '(', ')', true).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 1 }, around.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 8 }, around.b); const inside = pairRange(w, .{ .row = 0, .col = 3 }, '(', ')', false).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 2 }, inside.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 7 }, inside.b); // cursor on the nested open picks the nested pair const nested = pairRange(w, .{ .row = 0, .col = 5 }, '(', ')', false).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, nested.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, nested.b); // empty pair: no inside const empty = [_][]const u8{"()"}; try std.testing.expectEqual(@as(?Range, null), pairRange(&empty, .{ .row = 0, .col = 0 }, '(', ')', false)); // not enclosed try std.testing.expectEqual(@as(?Range, null), pairRange(&empty, .{ .row = 0, .col = 1 }, '[', ']', false)); } test "quoteRange line-scoped" { const lines = [_][]const u8{"say 'hi there' end"}; const w = &lines; const r = quoteRange(w, .{ .row = 0, .col = 7 }, '\'', true).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, r.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 13 }, r.b); const ri = quoteRange(w, .{ .row = 0, .col = 7 }, '\'', false).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, ri.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 12 }, ri.b); // cursor after the pair: not enclosed try std.testing.expectEqual(@as(?Range, null), quoteRange(w, .{ .row = 0, .col = 16 }, '\'', true)); } test "wordRange inside/around" { const lines = [_][]const u8{"one two.three"}; const w = &lines; const r = wordRange(w, .{ .row = 0, .col = 1 }, false, false).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, r.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 2 }, r.b); // around eats the trailing spaces const ra = wordRange(w, .{ .row = 0, .col = 1 }, false, true).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, ra.b); // long word spans the dot const rl = wordRange(w, .{ .row = 0, .col = 6 }, true, false).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, rl.a); try std.testing.expectEqual(Cursor{ .row = 0, .col = 13 }, rl.b); // on whitespace: none try std.testing.expectEqual(@as(?Range, null), wordRange(w, .{ .row = 0, .col = 3 }, false, false)); } test "paragraphRange inside/around" { const lines = [_][]const u8{ "a", "b", "", "c" }; const w = &lines; const r = paragraphRange(w, .{ .row = 1, .col = 0 }, false).?; try std.testing.expectEqual(Cursor{ .row = 0, .col = 0 }, r.a); try std.testing.expectEqual(Cursor{ .row = 1, .col = 0 }, r.b); const ra = paragraphRange(w, .{ .row = 1, .col = 0 }, true).?; try std.testing.expectEqual(Cursor{ .row = 2, .col = 0 }, ra.b); try std.testing.expectEqual(@as(?Range, null), paragraphRange(w, .{ .row = 2, .col = 0 }, false)); } test "advanceBy" { try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, advanceBy(.{ .row = 0, .col = 2 }, "abc")); try std.testing.expectEqual(Cursor{ .row = 2, .col = 1 }, advanceBy(.{ .row = 0, .col = 2 }, "a\nbc\nd")); } test "replaceRange" { const a = std.testing.allocator; const r = try replaceRange(a, "hello world", .{ .row = 0, .col = 0 }, .{ .row = 0, .col = 4 }, "bye"); defer a.free(r); try std.testing.expectEqualStrings("bye world", r); } test "replaceChars overwrites newlines too" { const a = std.testing.allocator; const r = try replaceChars(a, "ab\ncd", .{ .row = 0, .col = 1 }, .{ .row = 1, .col = 0 }, 'x'); defer a.free(r); try std.testing.expectEqualStrings("axxxd", r); } test "changeCase" { const a = std.testing.allocator; const t = try changeCase(a, "aB cD", .{ .row = 0, .col = 0 }, .{ .row = 0, .col = 4 }, .toggle); defer a.free(t); try std.testing.expectEqualStrings("Ab Cd", t); const lo = try changeCase(a, "AB CD", .{ .row = 0, .col = 0 }, .{ .row = 0, .col = 1 }, .lower); defer a.free(lo); try std.testing.expectEqualStrings("ab CD", lo); const up = try changeCase(a, "ab cd", .{ .row = 0, .col = 3 }, .{ .row = 0, .col = 4 }, .upper); defer a.free(up); try std.testing.expectEqualStrings("ab CD", up); } test "joinLine" { const a = std.testing.allocator; const r = (try joinLine(a, "one\n two\nthree", 0)).?; defer a.free(r.content); try std.testing.expectEqualStrings("one two\nthree", r.content); try std.testing.expectEqual(@as(usize, 3), r.col); // last line: nothing to join try std.testing.expectEqual(@as(?@TypeOf(r), null), try joinLine(a, "one", 0)); } test "indentLines add and remove" { const a = std.testing.allocator; const r = try indentLines(a, "one\n\ntwo", 0, 2, true); defer a.free(r); try std.testing.expectEqualStrings(" one\n\n two", r); const u = try indentLines(a, " one\n\ttwo\n three\nx", 0, 2, false); defer a.free(u); try std.testing.expectEqualStrings("one\ntwo\nthree\nx", u); } test "adjustNumber" { const a = std.testing.allocator; const r = (try adjustNumber(a, "x 41 y", .{ .row = 0, .col = 3 }, 1)).?; defer a.free(r.content); try std.testing.expectEqualStrings("x 42 y", r.content); try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, r.cur); // negative, cursor on the '-' const n = (try adjustNumber(a, "v=-1;", .{ .row = 0, .col = 2 }, -1)).?; defer a.free(n.content); try std.testing.expectEqualStrings("v=-2;", n.content); try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, n.cur); // width change moves the last-digit column const g = (try adjustNumber(a, "9", .{ .row = 0, .col = 0 }, 1)).?; defer a.free(g.content); try std.testing.expectEqualStrings("10", g.content); try std.testing.expectEqual(Cursor{ .row = 0, .col = 1 }, g.cur); // not on a number try std.testing.expectEqual(@as(?@TypeOf(r), null), try adjustNumber(a, "abc", .{ .row = 0, .col = 0 }, 1)); } test "deleteSpan including the newline" { const a = std.testing.allocator; const r = try deleteSpan(a, "hello world", .{ .row = 0, .col = 2 }, .{ .row = 0, .col = 5 }); defer a.free(r); try std.testing.expectEqualStrings("he world", r); const j = try deleteSpan(a, "ab\ncd", .{ .row = 0, .col = 2 }, .{ .row = 1, .col = 0 }); defer a.free(j); try std.testing.expectEqualStrings("abcd", j); // empty span: copy const e = try deleteSpan(a, "ab", .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 1 }); defer a.free(e); try std.testing.expectEqualStrings("ab", e); } test "pasteLineBelow" { const content = "one\ntwo"; const a = std.testing.allocator; const r = try pasteLineBelow(a, content, 0, "INSERTED"); defer a.free(r); try std.testing.expectEqualStrings("one\nINSERTED\ntwo", r); // paste below last line const r2 = try pasteLineBelow(a, content, 1, "END"); defer a.free(r2); try std.testing.expectEqualStrings("one\ntwo\nEND", r2); // multiline yanked text const r3 = try pasteLineBelow(a, content, 0, "a\nb"); defer a.free(r3); try std.testing.expectEqualStrings("one\na\nb\ntwo", r3); }