summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-18 12:09:12 -0300
committerGabriel Schneider <[email protected]>2026-08-18 23:45:25 -0300
commitc24a9e40215bc30c68c9db7675f1c6a07d9bbae3 (patch)
tree9afdb1f3d128bda3f3cf4837c2893982b0a8f69c /src
parent31fece62f56aa2311e2325de83659edbc9e641db (diff)
downloadpardes-c24a9e40215bc30c68c9db7675f1c6a07d9bbae3.tar.gz
pardes-c24a9e40215bc30c68c9db7675f1c6a07d9bbae3.zip
modal + file_pane: rework editing math and pane behavior, config/syntax additions, unit tests
Diffstat (limited to 'src')
-rw-r--r--src/config.zig7
-rw-r--r--src/file_pane.zig122
-rw-r--r--src/grammar_manifest.zig1
-rw-r--r--src/modal.zig418
-rw-r--r--src/normal_input.zig12
-rw-r--r--src/pardes.zig462
-rw-r--r--src/pdf_pane_integration_test.zig3
-rw-r--r--src/syntax.zig26
8 files changed, 772 insertions, 279 deletions
diff --git a/src/config.zig b/src/config.zig
index 9f18f75e..f58716e5 100644
--- a/src/config.zig
+++ b/src/config.zig
@@ -581,7 +581,10 @@ pub const MINH: u16 = 3;
/// definition of "the word under the cursor" for Look, Execute, the tag chord
/// and every search result row.
pub fn isFileChar(c: u8) bool {
- return std.ascii.isAlphanumeric(c) or switch (c) {
+ // Non-ASCII bytes belong to their UTF-8 word as a unit. Bounds are byte
+ // offsets, so accepting every high byte keeps Unicode paths/identifiers
+ // intact instead of returning a slice through one codepoint.
+ return c >= 0x80 or std.ascii.isAlphanumeric(c) or switch (c) {
'.', '-', '+', '/', ':', '@', '_', '~' => true,
else => false,
};
@@ -677,7 +680,7 @@ pub const image_exts = [_][]const u8{ ".png", ".jpg", ".jpeg", ".gif", ".bmp", "
/// token for them would be a divergence nothing asked for.
pub const comment_token_default = "#";
pub const comment_tokens: []const struct { exts: []const []const u8, token: []const u8 } = &.{
- .{ .token = "//", .exts = &.{ ".zig", ".zon", ".c", ".h", ".cpp", ".cc", ".cxx", ".hpp", ".hh", ".hxx", ".rs", ".go", ".java", ".scala", ".sc", ".kt", ".kts", ".cs", ".csx", ".php", ".pas", ".pp", ".p", ".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".swift", ".dart" } },
+ .{ .token = "//", .exts = &.{ ".zig", ".zon", ".c", ".h", ".cpp", ".cc", ".cxx", ".hpp", ".hh", ".hxx", ".rs", ".go", ".java", ".scala", ".sc", ".kt", ".kts", ".cs", ".csx", ".php", ".pas", ".pp", ".p", ".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".typ", ".typst", ".swift", ".dart" } },
.{ .token = "#", .exts = &.{ ".py", ".pyw", ".sh", ".bash", ".zsh", ".rb", ".rake", ".ex", ".exs", ".ps1", ".psm1", ".psd1", ".pl", ".pm", ".r", ".jl", ".nix", ".toml", ".yaml", ".yml", ".cmake", ".mk", ".tf" } },
.{ .token = "--", .exts = &.{ ".lua", ".hs", ".lhs", ".elm", ".sql", ".adb", ".ads", ".ada" } },
.{ .token = ";", .exts = &.{ ".clj", ".cljs", ".cljc", ".edn", ".el", ".lisp", ".scm", ".asm", ".s" } },
diff --git a/src/file_pane.zig b/src/file_pane.zig
index cf2b9571..abf37448 100644
--- a/src/file_pane.zig
+++ b/src/file_pane.zig
@@ -4,6 +4,7 @@
//! number gutter and the syntax recolor). The rest of a file pane's behaviour
//! is the pane machinery in pardes.zig, which does not care what kind it is.
const std = @import("std");
+const vaxis = @import("vaxis");
const pardes = @import("pardes.zig");
const config = @import("config.zig");
const Pardes = pardes.Pardes;
@@ -82,13 +83,23 @@ pub fn dumpPane(
};
}
+pub fn graphemeDisplayWidth(grapheme: []const u8) usize {
+ if (std.mem.eql(u8, grapheme, "\t")) return config.tab_width;
+ return @max(1, @as(usize, vaxis.gwidth.gwidth(grapheme, .unicode)));
+}
+
pub fn byteDisplayWidth(byte: u8) usize {
return if (byte == '\t') config.tab_width else 1;
}
pub fn displayWidth(text: []const u8) usize {
var width: usize = 0;
- for (text) |byte| width +|= byteDisplayWidth(byte);
+ var at: usize = 0;
+ while (at < text.len) {
+ const end = modal.nextGrapheme(text, at);
+ width +|= graphemeDisplayWidth(text[at..end]);
+ at = end;
+ }
return width;
}
@@ -96,10 +107,13 @@ pub fn displayWidth(text: []const u8) usize {
/// maps back to that one tab byte.
pub fn byteAtDisplay(text: []const u8, display_col: usize) usize {
var col: usize = 0;
- for (text, 0..) |byte, i| {
- const next = col +| byteDisplayWidth(byte);
- if (display_col < next) return i;
+ var at: usize = 0;
+ while (at < text.len) {
+ const end = modal.nextGrapheme(text, at);
+ const next = col +| graphemeDisplayWidth(text[at..end]);
+ if (display_col < next) return at;
col = next;
+ at = end;
}
return text.len;
}
@@ -107,8 +121,8 @@ pub fn byteAtDisplay(text: []const u8, display_col: usize) usize {
/// File cursor columns may live past EOL. Tabs expand before that boundary;
/// every virtual column after it remains one screen cell.
pub fn rawDisplayCol(line_text: []const u8, raw_col: usize) usize {
- const bounded = @min(raw_col, line_text.len);
- return displayWidth(line_text[0..bounded]) +| (raw_col - bounded);
+ const bounded = modal.graphemeStart(line_text, @min(raw_col, line_text.len));
+ return displayWidth(line_text[0..bounded]) +| (raw_col -| line_text.len);
}
pub fn rawAtDisplay(line_text: []const u8, display_col: usize) usize {
@@ -117,6 +131,27 @@ pub fn rawAtDisplay(line_text: []const u8, display_col: usize) usize {
return byteAtDisplay(line_text, display_col);
}
+pub fn byteAtDisplayFrom(line_text: []const u8, from_raw: usize, display_col: usize) usize {
+ if (from_raw >= line_text.len) return from_raw +| display_col;
+ const from = modal.graphemeStart(line_text, from_raw);
+ return from +| rawAtDisplay(line_text[from..], display_col);
+}
+
+pub fn lineDisplayOffset(line_text: []const u8, from_raw: usize, to_raw: usize) i32 {
+ const from_display = rawDisplayCol(line_text, from_raw);
+ const to_display = rawDisplayCol(line_text, to_raw);
+ if (to_display >= from_display) return @intCast(to_display - from_display);
+ return -@as(i32, @intCast(from_display - to_display));
+}
+
+pub fn lineDisplayEndOffset(line_text: []const u8, from_raw: usize, at_raw: usize) i32 {
+ const start = lineDisplayOffset(line_text, from_raw, at_raw);
+ if (at_raw >= line_text.len) return start;
+ const at = modal.graphemeStart(line_text, at_raw);
+ const end = modal.nextGrapheme(line_text, at);
+ return start + @as(i32, @intCast(graphemeDisplayWidth(line_text[at..end]))) - 1;
+}
+
pub fn sourceLine(pane: *const Pane, row: i32) []const u8 {
const f = pane.file orelse return "";
if (row < 0) return "";
@@ -127,51 +162,60 @@ pub fn displayOffset(pane: *const Pane, row: i32, from_raw: i32, to_raw: i32) i3
const line_text = sourceLine(pane, row);
const from: usize = @intCast(@max(0, from_raw));
const to: usize = @intCast(@max(0, to_raw));
- const from_display = rawDisplayCol(line_text, from);
- const to_display = rawDisplayCol(line_text, to);
- if (to_display >= from_display) return @intCast(to_display - from_display);
- return -@as(i32, @intCast(from_display - to_display));
+ return lineDisplayOffset(line_text, from, to);
}
pub fn displayEndOffset(pane: *const Pane, row: i32, from_raw: i32, at_raw: i32) i32 {
- const start = displayOffset(pane, row, from_raw, at_raw);
const line_text = sourceLine(pane, row);
- const at: usize = @intCast(@max(0, at_raw));
- if (at >= line_text.len) return start;
- return start + @as(i32, @intCast(byteDisplayWidth(line_text[at]))) - 1;
+ return lineDisplayEndOffset(line_text, @intCast(@max(0, from_raw)), @intCast(@max(0, at_raw)));
}
pub fn byteAtRowDisplay(pane: *const Pane, row: i32, from_raw: i32, display_col: i32) i32 {
const line_text = sourceLine(pane, row);
- const from: usize = @intCast(@max(0, from_raw));
- const display: usize = @intCast(@max(0, display_col));
- if (from >= line_text.len) return @intCast(from +| display);
- return @intCast(from +| rawAtDisplay(line_text[from..], display));
+ return @intCast(byteAtDisplayFrom(line_text, @intCast(@max(0, from_raw)), @intCast(@max(0, display_col))));
}
-/// Convert between rendered pane rows (whose file body includes PREFIX_W)
-/// and source-byte columns. Non-file rows are already in byte coordinates.
+/// Convert between rendered cells and UTF-8 byte columns. Tag rows always
+/// need grapheme conversion; file body rows additionally skip PREFIX_W.
pub fn renderedLineByteCol(pane: *const Pane, row: i32, line_text: []const u8, display_col: usize) usize {
- if (pane.file == null or row < pardes.BOX_H) return display_col;
+ if (row < pardes.BOX_H) return rawAtDisplay(line_text, display_col);
+ if (pane.file == null) return rawAtDisplay(line_text, display_col);
const prefix = @min(@as(usize, config.PREFIX_W), line_text.len);
if (display_col <= prefix) return display_col;
return prefix +| rawAtDisplay(line_text[prefix..], display_col - prefix);
}
pub fn renderedLineDisplayCol(pane: *const Pane, row: i32, line_text: []const u8, byte_col: usize) usize {
- if (pane.file == null or row < pardes.BOX_H) return byte_col;
+ if (row < pardes.BOX_H) return rawDisplayCol(line_text, byte_col);
+ if (pane.file == null) return rawDisplayCol(line_text, byte_col);
const prefix = @min(@as(usize, config.PREFIX_W), line_text.len);
if (byte_col <= prefix) return byte_col;
return prefix +| rawDisplayCol(line_text[prefix..], byte_col - prefix);
}
+test "display columns map complete Unicode graphemes" {
+ const text = "é界e\u{301}x";
+ try std.testing.expectEqual(@as(usize, 5), displayWidth(text));
+ try std.testing.expectEqual(@as(usize, 0), byteAtDisplay(text, 0));
+ try std.testing.expectEqual(@as(usize, 2), byteAtDisplay(text, 1));
+ try std.testing.expectEqual(@as(usize, 2), byteAtDisplay(text, 2));
+ try std.testing.expectEqual(@as(usize, 5), byteAtDisplay(text, 3));
+ try std.testing.expectEqual(@as(usize, 8), byteAtDisplay(text, 4));
+ try std.testing.expectEqual(text.len, byteAtDisplay(text, 5));
+ try std.testing.expectEqual(@as(usize, 3), rawDisplayCol(text, 5));
+ try std.testing.expectEqual(@as(usize, 5), rawAtDisplay(text, 3));
+ try std.testing.expectEqual(@as(usize, 2), graphemeDisplayWidth("👩\u{200d}🚀"));
+}
+
fn fitEnd(text: []const u8, start: usize, width: usize) usize {
var end = start;
var used: usize = 0;
- while (end < text.len) : (end += 1) {
- const next = used +| byteDisplayWidth(text[end]);
- if (next > width) return if (end == start) start + 1 else end;
- used = next;
+ while (end < text.len) {
+ const next_end = modal.nextGrapheme(text, end);
+ const next_used = used +| graphemeDisplayWidth(text[end..next_end]);
+ if (next_used > width) return if (end == start) next_end else end;
+ used = next_used;
+ end = next_end;
}
return end;
}
@@ -253,7 +297,7 @@ pub fn textOffset(pane: *Pane, text: []const u8, cursor: modal.Cursor) usize {
const row = @min(cursor.row, index.len - 1);
const start = index[row];
const end = if (row + 1 < index.len) index[row + 1] - 1 else text.len;
- return @min(start + cursor.col, end);
+ return start + modal.graphemeStart(text[start..end], @min(cursor.col, end - start));
}
pub fn textLineStart(pane: *Pane, text: []const u8, row: usize) usize {
@@ -274,7 +318,9 @@ pub fn textPosition(pane: *Pane, text: []const u8, offset: usize) modal.Cursor {
return std.math.order(key, item);
}
}.cmp) - 1;
- return .{ .row = row, .col = bounded - index[row] };
+ const start = index[row];
+ const end = if (row + 1 < index.len) index[row + 1] - 1 else text.len;
+ return .{ .row = row, .col = modal.graphemeStart(text[start..end], @min(bounded - start, end - start)) };
}
pub fn open(p: *Pardes, id: usize, path: []const u8, line: usize) !*Pane {
@@ -577,17 +623,14 @@ fn fillBody(dst: ?[]u8, pane: *Pane, f: *State, width: usize, record_wrap: bool)
if (dst) |out| @memcpy(out[written..][0..prefix.len], prefix);
written += prefix.len;
- // A byte cut, like the hscroll one below, and it can land inside a
- // multi-byte glyph for the same reason: columns here are BYTES.
- // Surface.print decodes by hand and emits U+FFFD per undecodable
- // byte, so a split glyph renders as a replacement char rather than
- // panicking — see test/snapshots/badutf.snap. It cannot overflow
- // the pane either: a UTF-8 sequence is never fewer bytes than the
- // cells it draws in.
+ // Wrap and horizontal-scroll cuts are always grapheme boundaries.
+ // Source columns remain byte offsets, while widths are terminal
+ // cells; keeping the conversion here prevents a view operation
+ // from manufacturing malformed UTF-8.
const end = if (width == 0) text.len else fitEnd(text, at, width);
const take = end - at;
const cut = if (pane.hscroll > 0 and width == 0)
- @min(@as(usize, @intCast(pane.hscroll)), take)
+ modal.graphemeStart(text[at..end], @min(@as(usize, @intCast(pane.hscroll)), take))
else
0;
const shown = text[at + cut .. end];
@@ -690,10 +733,12 @@ pub fn recolorSyntax(p: *Pardes, pane: *Pane, f: *State, r: pardes.Rect, tx: u16
else
line.len;
}
+ hs = modal.graphemeStart(line, @min(hs, line.len));
var c: usize = 0;
var screen_c: usize = 0;
- while (hs + c < limit and config.PREFIX_W + screen_c < tw) : (c += 1) {
- const cells = byteDisplayWidth(line[hs + c]);
+ while (hs + c < limit and config.PREFIX_W + screen_c < tw) {
+ const grapheme_end = @min(limit, modal.nextGrapheme(line, hs + c));
+ const cells = graphemeDisplayWidth(line[hs + c .. grapheme_end]);
const idx = base + hs + c;
if (idx >= f.highlight_start) {
const hidx = idx - f.highlight_start;
@@ -710,6 +755,7 @@ pub fn recolorSyntax(p: *Pardes, pane: *Pane, f: *State, r: pardes.Rect, tx: u16
}
}
screen_c += cells;
+ c = grapheme_end - hs;
}
}
}
diff --git a/src/grammar_manifest.zig b/src/grammar_manifest.zig
index bd89e0a0..101cd6f6 100644
--- a/src/grammar_manifest.zig
+++ b/src/grammar_manifest.zig
@@ -40,5 +40,6 @@ pub const all = [_]Grammar{
.{ .name = "ruby", .dep = "ts_ruby", .exts = &.{ ".rb", ".rake" }, .tier = .full, .scanner = true },
.{ .name = "rust", .dep = "ts_rust", .exts = &.{".rs"}, .tier = .full, .scanner = true },
.{ .name = "scala", .dep = "ts_scala", .exts = &.{ ".scala", ".sc" }, .tier = .full, .scanner = true },
+ .{ .name = "typst", .dep = "ts_typst", .exts = &.{ ".typ", ".typst" }, .tier = .full, .scanner = true, .query = "queries/typst/highlights.scm" },
.{ .name = "zig", .dep = "ts_zig", .exts = &.{ ".zig", ".zon" }, .tier = .zig },
};
diff --git a/src/modal.zig b/src/modal.zig
index 6ab7ee35..a98cfeb4 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -1,12 +1,14 @@
const std = @import("std");
+const uucode = @import("uucode");
// Modal-editing text math, kept free of vaxis/ghostty so it can be unit-tested
// in isolation (see the `unit-test` build step). main.zig wires this onto the
// pane's cursor + (for file panes) its content.
//
-// The cursor sits ON a character: col is a char index in [0, line.len]; col ==
-// line.len means "on the line terminator / after the last char". Motions are
-// written to land on real characters; main.zig clamps for display.
+// The cursor sits ON a grapheme: col is its UTF-8 byte offset in
+// [0, line.len]; col == line.len means "on the line terminator / after the
+// last grapheme". Motions never leave a cursor in the middle of UTF-8 or an
+// extended grapheme cluster.
pub const Cursor = struct {
row: usize = 0,
@@ -21,23 +23,57 @@ pub const Cursor = struct {
// non-ws, ws = space/tab/newline).
pub const Kind = enum { word, punct, ws };
+fn codepointAt(text: []const u8, off: usize) u21 {
+ if (off >= text.len) return 0xFFFD;
+ const n = std.unicode.utf8ByteSequenceLength(text[off]) catch return 0xFFFD;
+ if (off + n > text.len) return 0xFFFD;
+ return std.unicode.utf8Decode(text[off .. off + n]) catch 0xFFFD;
+}
+
+fn isUnicodeWhitespace(cp: u21) bool {
+ if (cp == ' ' or (cp >= '\t' and cp <= '\r') or cp == 0x85) return true;
+ return switch (uucode.get(.general_category, cp)) {
+ .separator_space, .separator_line, .separator_paragraph => true,
+ else => false,
+ };
+}
+
+fn kindOfCodepoint(cp: u21) Kind {
+ if (isUnicodeWhitespace(cp)) return .ws;
+ if (cp == '_') return .word;
+ return switch (uucode.get(.general_category, cp)) {
+ .letter_uppercase,
+ .letter_lowercase,
+ .letter_titlecase,
+ .letter_modifier,
+ .letter_other,
+ .mark_nonspacing,
+ .mark_spacing_combining,
+ .mark_enclosing,
+ .number_decimal_digit,
+ .number_letter,
+ .number_other,
+ .punctuation_connector,
+ => .word,
+ else => .punct,
+ };
+}
+
pub fn kindOf(c: u8) Kind {
- if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws;
- if (std.ascii.isAlphanumeric(c) or c == '_') return .word;
- return .punct;
+ return kindOfCodepoint(c);
}
// "long word" (W/B/E): only whitespace separates; punct is part of a word.
-fn kindOfLong(c: u8) Kind {
- if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws;
- return .word;
+fn kindOfLong(cp: u21) Kind {
+ return if (isUnicodeWhitespace(cp)) .ws else .word;
}
fn kindAt(lines: []const []const u8, c: Cursor, long: bool) Kind {
if (c.row >= lines.len) return .ws;
const line = lines[c.row];
if (c.col >= line.len) return .ws; // line terminator / EOF = whitespace
- return if (long) kindOfLong(line[c.col]) else kindOf(line[c.col]);
+ const cp = codepointAt(line, graphemeStart(line, c.col));
+ return if (long) kindOfLong(cp) else kindOfCodepoint(cp);
}
fn lineLenOf(lines: []const []const u8, row: usize) usize {
@@ -51,7 +87,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool {
if (c.row >= lines.len) return false;
const llen = lineLenOf(lines, c.row);
if (c.col < llen) {
- c.col += 1;
+ c.col = nextGrapheme(lines[c.row], c.col);
return true;
}
// at the newline: move to next line start
@@ -65,7 +101,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool {
fn stepBwd(lines: []const []const u8, c: *Cursor) bool {
if (c.col > 0) {
- c.col -= 1;
+ c.col = prevGrapheme(lines[c.row], c.col);
return true;
}
if (c.row == 0) return false;
@@ -83,7 +119,7 @@ fn atEof(lines: []const []const u8, c: Cursor) bool {
pub fn firstNonWs(line: []const u8) usize {
var i: usize = 0;
- while (i < line.len and (line[i] == ' ' or line[i] == '\t')) i += 1;
+ while (i < line.len and isUnicodeWhitespace(codepointAt(line, i))) i = nextGrapheme(line, i);
return i;
}
@@ -95,7 +131,7 @@ pub fn lineStart(c: Cursor) Cursor {
pub fn lineEnd(lines: []const []const u8, c: Cursor) Cursor {
const llen = lineLenOf(lines, c.row);
- return .{ .row = c.row, .col = if (llen == 0) 0 else llen - 1 };
+ return .{ .row = c.row, .col = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen) };
}
pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor {
@@ -107,28 +143,33 @@ pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor {
// ---- char/line motions ----
-pub fn charLeft(c: Cursor) Cursor {
- return .{ .row = c.row, .col = if (c.col > 0) c.col - 1 else 0 };
+pub fn charLeft(lines: []const []const u8, c: Cursor) Cursor {
+ if (c.row >= lines.len) return .{ .row = c.row, .col = 0 };
+ return .{ .row = c.row, .col = prevGrapheme(lines[c.row], @min(c.col, lines[c.row].len)) };
}
pub fn charRight(lines: []const []const u8, c: Cursor) Cursor {
const llen = lineLenOf(lines, c.row);
- const last = if (llen == 0) 0 else llen - 1;
- return .{ .row = c.row, .col = if (c.col < last) c.col + 1 else last };
+ const last = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen);
+ return .{ .row = c.row, .col = if (c.col < last) @min(nextGrapheme(lines[c.row], c.col), last) else last };
+}
+
+fn clampLineCol(line: []const u8, col: usize) usize {
+ if (line.len == 0) return 0;
+ const last = prevGrapheme(line, line.len);
+ return graphemeStart(line, @min(col, last));
}
pub fn lineDown(lines: []const []const u8, c: Cursor) Cursor {
const nr = if (c.row + 1 < lines.len) c.row + 1 else c.row;
const llen = lineLenOf(lines, nr);
- const last = if (llen == 0) 0 else llen - 1;
- return .{ .row = nr, .col = if (c.col < last) c.col else last };
+ return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) };
}
pub fn lineUp(lines: []const []const u8, c: Cursor) Cursor {
const nr = if (c.row > 0) c.row - 1 else c.row;
const llen = lineLenOf(lines, nr);
- const last = if (llen == 0) 0 else llen - 1;
- return .{ .row = nr, .col = if (c.col < last) c.col else last };
+ return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) };
}
// ---- word motions ----
@@ -205,19 +246,22 @@ pub fn gotoLast(lines: []const []const u8) Cursor {
}
// the character the cursor sits on; line terminators / EOF read as '\n'.
-fn charAt(lines: []const []const u8, c: Cursor) u8 {
+fn codepointAtCursor(lines: []const []const u8, c: Cursor) u21 {
if (c.row >= lines.len) return '\n';
const line = lines[c.row];
if (c.col >= line.len) return '\n';
- return line[c.col];
+ return codepointAt(line, graphemeStart(line, c.col));
+}
+
+fn charAt(lines: []const []const u8, c: Cursor) u8 {
+ const cp = codepointAtCursor(lines, c);
+ return if (cp <= 0x7f) @intCast(cp) else 0;
}
// `f`/`F`/`t`/`T`: the nth occurrence of `ch` after/before the cursor, across
// line boundaries (helix: not confined to the line). `till` stops one position
// short of the hit. Returns null (no move) when there aren't n occurrences.
pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: bool, n: usize) ?Cursor {
- if (ch > 0x7f) return null; // ponytail: ASCII targets only (byte columns)
- const target: u8 = @intCast(ch);
var p = clampToChar(lines, c);
var left = if (n == 0) 1 else n;
while (left > 0) {
@@ -226,7 +270,7 @@ pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till:
} else {
if (!stepBwd(lines, &p)) return null;
}
- if (charAt(lines, p) == target) left -= 1;
+ if (codepointAtCursor(lines, p) == ch) left -= 1;
}
if (till) {
if (fwd) _ = stepBwd(lines, &p) else _ = stepFwd(lines, &p);
@@ -369,16 +413,32 @@ pub fn wordRange(lines: []const []const u8, c0: Cursor, long: bool, around: bool
const k = kindAt(lines, c, long);
if (k == .ws) return null;
var lo = c.col;
- while (lo > 0 and kindAt(lines, .{ .row = c.row, .col = lo - 1 }, long) == k) lo -= 1;
+ while (lo > 0) {
+ const prev = prevGrapheme(line, lo);
+ if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != k) break;
+ lo = prev;
+ }
var hi = c.col;
- while (hi + 1 < line.len and kindAt(lines, .{ .row = c.row, .col = hi + 1 }, long) == k) hi += 1;
+ while (true) {
+ const next = nextGrapheme(line, hi);
+ if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != k) break;
+ hi = next;
+ }
if (around) {
var h2 = hi;
- while (h2 + 1 < line.len and (line[h2 + 1] == ' ' or line[h2 + 1] == '\t')) h2 += 1;
+ while (true) {
+ const next = nextGrapheme(line, h2);
+ if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != .ws) break;
+ h2 = next;
+ }
if (h2 != hi) {
hi = h2;
} else {
- while (lo > 0 and (line[lo - 1] == ' ' or line[lo - 1] == '\t')) lo -= 1;
+ while (lo > 0) {
+ const prev = prevGrapheme(line, lo);
+ if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != .ws) break;
+ lo = prev;
+ }
}
}
return .{ .a = .{ .row = c.row, .col = lo }, .b = .{ .row = c.row, .col = hi } };
@@ -403,7 +463,7 @@ pub fn paragraphRange(lines: []const []const u8, c0: Cursor, around: bool) ?Rang
}
}
const llen = lineLenOf(lines, r1);
- return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else llen - 1 } };
+ return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else prevGrapheme(lines[r1], llen) } };
}
// ---- helpers used by motions + main.zig ----
@@ -415,7 +475,7 @@ pub fn clampToChar(lines: []const []const u8, c: Cursor) Cursor {
}
const llen = lineLenOf(lines, c.row);
if (llen == 0) return .{ .row = c.row, .col = 0 };
- return .{ .row = c.row, .col = @min(c.col, llen - 1) };
+ return .{ .row = c.row, .col = clampLineCol(lines[c.row], c.col) };
}
pub fn lineCount(content: []const u8) usize {
@@ -465,8 +525,9 @@ pub fn insertAt(alloc: std.mem.Allocator, content: []const u8, c: Cursor, text:
pub fn deleteChar(alloc: std.mem.Allocator, content: []const u8, c: Cursor) ![]u8 {
const line = lineSlice(content, c.row);
if (c.col >= line.len) return alloc.dupe(u8, content);
- const off = lineStartOffset(content, c.row) + c.col;
- return spliceAlloc(alloc, content, off, off + 1, "");
+ const col = graphemeStart(line, c.col);
+ const line_start = lineStartOffset(content, c.row);
+ return spliceAlloc(alloc, content, line_start + col, line_start + nextGrapheme(line, col), "");
}
// delete whole lines [r0, r1] inclusive (the line content + their terminators).
@@ -520,9 +581,13 @@ fn rangeBytes(content: []const u8, a: Cursor, b: Cursor) struct { s: usize, e: u
lo = b;
hi = a;
}
- const s = lineStartOffset(content, lo.row) + @min(lo.col, lineSlice(content, lo.row).len);
- var e = lineStartOffset(content, hi.row) + @min(hi.col, lineSlice(content, hi.row).len);
- if (e < content.len) e += 1; // include the char under the head
+ const lo_line = lineSlice(content, lo.row);
+ const hi_line = lineSlice(content, hi.row);
+ const lo_col = graphemeStart(lo_line, @min(lo.col, lo_line.len));
+ const hi_col = graphemeStart(hi_line, @min(hi.col, hi_line.len));
+ const s = lineStartOffset(content, lo.row) + lo_col;
+ var e = lineStartOffset(content, hi.row) + hi_col;
+ if (e < content.len) e = nextGrapheme(content, e); // include the grapheme under the head
return .{ .s = s, .e = @max(s, e) };
}
@@ -593,10 +658,22 @@ pub fn replaceRange(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b:
// `r<ch>`: overwrite every char in the inclusive range [a, b] with `ch` —
// NEWLINES TOO (helix replace maps every grapheme, so `xrz` joins lines).
-pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u8) ![]u8 {
+pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u21) ![]u8 {
const r = rangeBytes(content, a, b);
- const out = try alloc.dupe(u8, content);
- for (out[r.s..r.e]) |*p| p.* = ch;
+ var encoded: [4]u8 = undefined;
+ const encoded_len = try std.unicode.utf8Encode(ch, &encoded);
+ var count: usize = 0;
+ var at = r.s;
+ while (at < r.e) : (count += 1) at = nextGrapheme(content, at);
+ const replacement_len = count * encoded_len;
+ const out = try alloc.alloc(u8, content.len - (r.e - r.s) + replacement_len);
+ @memcpy(out[0..r.s], content[0..r.s]);
+ var write = r.s;
+ for (0..count) |_| {
+ @memcpy(out[write..][0..encoded_len], encoded[0..encoded_len]);
+ write += encoded_len;
+ }
+ @memcpy(out[write..], content[r.e..]);
return out;
}
@@ -729,9 +806,9 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C
//
// Gap-offset ranges over the FLAT buffer, ported faithfully from
// helix-core/src/movement.rs + selection.rs @ 278b24389 (the genizah
-// checkout). Positions are gap offsets 0..=text.len — "char indices" in
-// helix terms, bytes here (ASCII-exact; UTF-8 stepped by sequence, matching
-// the byte-column convention of the rest of pardes). A range with
+// checkout). Positions are UTF-8 gap offsets 0..=text.len. Stored columns
+// remain byte offsets, but every range boundary is an extended-grapheme
+// boundary. A range with
// head > anchor selects [anchor, head) with the block cursor ON head-1;
// head < anchor selects [head, anchor) with the cursor ON head. The
// differential suite (test/hxcases, `zig build hxdiff`) pins every behavior
@@ -739,19 +816,58 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C
pub const HxRange = struct { anchor: usize, head: usize };
-/// one grapheme forward (UTF-8 sequence step), clamped at text.len
+/// The grapheme containing `off`, or text.len at EOF. This is also the repair
+/// path for stale/external byte columns that happen to point into UTF-8.
+pub fn graphemeStart(text: []const u8, off: usize) usize {
+ const bounded = @min(off, text.len);
+ if (bounded == text.len) return text.len;
+ var it = uucode.grapheme.utf8Iterator(text);
+ while (it.nextGrapheme()) |g| {
+ if (bounded < g.end) return g.start;
+ }
+ return text.len;
+}
+
+/// one extended grapheme forward, clamped at text.len
pub fn nextGrapheme(text: []const u8, off: usize) usize {
if (off >= text.len) return text.len;
- var o = off + 1;
- while (o < text.len and (text[o] & 0xC0) == 0x80) o += 1;
- return o;
+ // The editor's own offsets are already boundaries. Keep the overwhelmingly
+ // common ASCII path O(1); only repair a continuation-byte input here.
+ if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1;
+ var start = off;
+ while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1;
+ if (start != off) start = graphemeStart(text, off);
+ var it = uucode.grapheme.utf8Iterator(text[start..]);
+ const g = it.nextGrapheme() orelse return @min(start + 1, text.len);
+ return start + g.end;
}
pub fn prevGrapheme(text: []const u8, off: usize) usize {
- if (off == 0) return 0;
- var o = off - 1;
- while (o > 0 and (text[o] & 0xC0) == 0x80) o -= 1;
- return o;
+ var bounded = @min(off, text.len);
+ if (bounded < text.len) {
+ const repaired = graphemeStart(text, bounded);
+ if (repaired < bounded) return repaired;
+ bounded = repaired;
+ }
+ if (bounded == 0) return 0;
+ if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1;
+ // Graphemes cannot cross a line break. Restrict the forward segmentation
+ // needed for a reverse step to the current line instead of rescanning the
+ // complete buffer.
+ const line_start = if (std.mem.lastIndexOfScalar(u8, text[0 .. bounded - 1], '\n')) |nl| nl + 1 else 0;
+ if (line_start == bounded) return bounded - 1; // the newline is its own editor cell
+ var it = uucode.grapheme.utf8Iterator(text[line_start..bounded]);
+ var last = line_start;
+ while (it.nextGrapheme()) |g| last = line_start + g.start;
+ return last;
+}
+
+/// Byte offset of the zero-based grapheme column, clamped to the line end.
+pub fn graphemeAtColumn(text: []const u8, column: usize) usize {
+ var at: usize = 0;
+ var col: usize = 0;
+ while (at < text.len and col < column) : (col += 1) at = nextGrapheme(text, at);
+ return at;
}
/// the block cursor cell of a range (helix Range::cursor)
@@ -792,14 +908,16 @@ pub fn hxLineEndIdx(text: []const u8, line: usize) usize {
/// gap offset -> (row, col) cell
pub fn hxPos(text: []const u8, off: usize) Cursor {
- const o = @min(off, text.len);
+ const bounded = @min(off, text.len);
// The line start is the byte after the last '\n' BEFORE off, which is the
// same number lineStartOffset(text, row) walks the whole prefix to reach —
// one backward scan of a single line instead of a second pass over
// everything above the cursor. On a multi-MB buffer that second pass was
// most of what a keystroke cost.
- const s = if (std.mem.lastIndexOfScalar(u8, text[0..o], '\n')) |nl| nl + 1 else 0;
- return .{ .row = hxLineOf(text, off), .col = o - s };
+ const s = if (std.mem.lastIndexOfScalar(u8, text[0..bounded], '\n')) |nl| nl + 1 else 0;
+ const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len;
+ const col = graphemeStart(text[s..e], @min(bounded - s, e - s));
+ return .{ .row = hxLineOf(text, s + col), .col = col };
}
/// (row, col) -> clamped gap offset; col == line length lands ON the '\n'
@@ -809,7 +927,8 @@ pub fn hxOff(text: []const u8, c: Cursor) usize {
// hxLineEndIdx(text, row) inlined: it starts by walking to `row` again,
// and we are already standing there
const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len;
- return @min(s + c.col, e);
+ const raw = @min(c.col, e - s);
+ return s + graphemeStart(text[s..e], raw);
}
pub const WordTarget = enum {
@@ -827,35 +946,36 @@ pub const WordTarget = enum {
// that distinction is load-bearing in reached_target.
const HxCat = enum { word, punct, ws, eol };
-fn hxCat(b: u8) HxCat {
- if (b == '\n' or b == '\r') return .eol;
- if (b == ' ' or b == '\t' or b == 0x0b or b == 0x0c) return .ws;
- if (std.ascii.isAlphanumeric(b) or b == '_' or b >= 0x80) return .word;
- return .punct;
+fn hxCatAt(text: []const u8, off: usize) HxCat {
+ if (off >= text.len) return .eol;
+ const cp = codepointAt(text, off);
+ if (cp == '\n' or cp == '\r') return .eol;
+ return switch (kindOfCodepoint(cp)) {
+ .word => .word,
+ .punct => .punct,
+ .ws => .ws,
+ };
}
-fn hxIsWs(b: u8) bool { // Rust char::is_whitespace (includes line endings)
- const c = hxCat(b);
+fn hxIsWs(c: HxCat) bool { // Rust char::is_whitespace (includes line endings)
return c == .ws or c == .eol;
}
-fn hxIsWordBoundary(a: u8, b: u8) bool {
- return hxCat(a) != hxCat(b);
+fn hxIsWordBoundary(a: HxCat, b: HxCat) bool {
+ return a != b;
}
-fn hxIsLongBoundary(a: u8, b: u8) bool {
- const ca = hxCat(a);
- const cb = hxCat(b);
- if ((ca == .word and cb == .punct) or (ca == .punct and cb == .word)) return false;
- return ca != cb;
+fn hxIsLongBoundary(a: HxCat, b: HxCat) bool {
+ if ((a == .word and b == .punct) or (a == .punct and b == .word)) return false;
+ return a != b;
}
-fn hxReached(target: WordTarget, prev: u8, next: u8) bool {
+fn hxReached(target: WordTarget, prev: HxCat, next: HxCat) bool {
return switch (target) {
- .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)),
- .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol),
- .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)),
- .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol),
+ .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (next == .eol or !hxIsWs(next)),
+ .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or next == .eol),
+ .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (next == .eol or !hxIsWs(next)),
+ .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or next == .eol),
};
}
@@ -896,45 +1016,35 @@ fn hxRangeToTarget(text: []const u8, target: WordTarget, origin: HxRange, is_pre
var anchor = origin.anchor;
var head = origin.head;
var it = origin.head;
- var prev_ch: ?u8 = if (is_prev)
- (if (it < text.len) text[it] else null)
+ var prev_cat: ?HxCat = if (is_prev)
+ (if (it < text.len) hxCatAt(text, it) else null)
else
- (if (it > 0) text[it - 1] else null);
+ (if (it > 0) hxCatAt(text, prevGrapheme(text, it)) else null);
// skip any initial newline characters
while (true) {
- const ch: u8 = if (is_prev) blk: {
- if (it == 0) break;
- break :blk text[it - 1];
- } else blk: {
- if (it >= text.len) break;
- break :blk text[it];
- };
- if (ch != '\n' and ch != '\r') break;
- if (is_prev) it -= 1 else it += 1;
- prev_ch = ch;
- if (is_prev) head -|= 1 else head += 1;
+ if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break;
+ const cell = if (is_prev) prevGrapheme(text, it) else it;
+ const cat = hxCatAt(text, cell);
+ if (cat != .eol) break;
+ it = if (is_prev) cell else nextGrapheme(text, cell);
+ prev_cat = cat;
+ head = it;
}
- if (prev_ch != null and hxCat(prev_ch.?) == .eol) anchor = head;
+ if (prev_cat == .eol) anchor = head;
// find the target position
const head_start = head;
while (true) {
- const next_ch: u8 = if (is_prev) blk: {
- if (it == 0) break;
- it -= 1;
- break :blk text[it];
- } else blk: {
- if (it >= text.len) break;
- const c = text[it];
- it += 1;
- break :blk c;
- };
- if (prev_ch == null or hxReached(target, prev_ch.?, next_ch)) {
+ if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break;
+ const cell = if (is_prev) prevGrapheme(text, it) else it;
+ const next_cat = hxCatAt(text, cell);
+ if (prev_cat == null or hxReached(target, prev_cat.?, next_cat)) {
if (head == head_start) anchor = head else break;
}
- prev_ch = next_ch;
- if (is_prev) head -|= 1 else head += 1;
+ prev_cat = next_cat;
+ it = if (is_prev) cell else nextGrapheme(text, cell);
+ head = it;
}
return .{ .anchor = anchor, .head = head };
}
@@ -1007,31 +1117,31 @@ pub fn hxVertTarget(text: []const u8, pos: usize, down: bool, count: usize, goal
const s = lineStartOffset(text, nline);
// hxLineEndIdx(text, nline) without its second walk to nline (see hxOff)
const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len;
- return @min(s + goal_col, e);
+ return s + graphemeStart(text[s..e], @min(goal_col, e - s));
}
/// f/F/t/T target cell. helix find_char: the exclusive (till) search starts
/// one further out so repeats make progress; not-found = null (no move).
-pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bool, count: usize) ?usize {
+pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u21, fwd: bool, till: bool, count: usize) ?usize {
var left = @max(1, count);
if (fwd) {
const head = nextGrapheme(text, cursor);
- var i = if (till) head + 1 else head;
+ var i = if (till) nextGrapheme(text, head) else head;
if (i > text.len) return null;
- while (i < text.len) : (i += 1) {
- if (text[i] == ch) {
+ while (i < text.len) : (i = nextGrapheme(text, i)) {
+ if (codepointAt(text, i) == ch) {
left -= 1;
- if (left == 0) return if (till) i - 1 else i;
+ if (left == 0) return if (till) prevGrapheme(text, i) else i;
}
}
return null;
}
- var i = if (till) cursor -| 1 else cursor;
+ var i = if (till) prevGrapheme(text, cursor) else cursor;
while (i > 0) {
- i -= 1;
- if (text[i] == ch) {
+ i = prevGrapheme(text, i);
+ if (codepointAt(text, i) == ch) {
left -= 1;
- if (left == 0) return if (till) i + 1 else i;
+ if (left == 0) return if (till) nextGrapheme(text, i) else i;
}
}
return null;
@@ -1040,26 +1150,19 @@ pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bo
// helix textobject.rs find_word_boundary
fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usize {
var prev: HxCat = if (fwd)
- (if (pos0 == 0) .ws else hxCat(text[pos0 - 1]))
+ (if (pos0 == 0) .ws else hxCatAt(text, prevGrapheme(text, pos0)))
else
- (if (pos0 >= text.len) .ws else hxCat(text[pos0]));
+ (if (pos0 >= text.len) .ws else hxCatAt(text, pos0));
var pos = pos0;
var it = pos0;
while (true) {
- const ch: u8 = if (fwd) blk: {
- if (it >= text.len) break;
- const c = text[it];
- it += 1;
- break :blk c;
- } else blk: {
- if (it == 0) break;
- it -= 1;
- break :blk text[it];
- };
- const cat = hxCat(ch);
+ if ((fwd and it >= text.len) or (!fwd and it == 0)) break;
+ const cell = if (fwd) it else prevGrapheme(text, it);
+ const cat = hxCatAt(text, cell);
if (cat == .eol or cat == .ws) return pos;
if (!long and cat != prev and pos != 0 and pos != text.len) return pos;
- if (fwd) pos += 1 else pos -|= 1;
+ it = if (fwd) nextGrapheme(text, cell) else cell;
+ pos = it;
prev = cat;
}
return pos;
@@ -1070,14 +1173,20 @@ fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usiz
pub fn hxTextobjectWord(text: []const u8, r: HxRange, around: bool, long: bool) HxRange {
const pos = hxCursor(text, r);
const word_start = hxFindWordBoundary(text, pos, false, long);
- const cat: HxCat = if (pos < text.len) hxCat(text[pos]) else .ws;
- const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, pos + 1, true, long);
+ const cat: HxCat = if (pos < text.len) hxCatAt(text, pos) else .ws;
+ const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, nextGrapheme(text, pos), true, long);
if (word_start == word_end or !around) return .{ .anchor = word_start, .head = word_end };
var end = word_end;
- while (end < text.len and hxIsWs(text[end]) and hxCat(text[end]) != .eol) end += 1;
+ while (end < text.len and hxIsWs(hxCatAt(text, end)) and hxCatAt(text, end) != .eol)
+ end = nextGrapheme(text, end);
if (end > word_end) return .{ .anchor = word_start, .head = end };
var start = word_start;
- while (start > 0 and hxIsWs(text[start - 1]) and hxCat(text[start - 1]) != .eol) start -= 1;
+ while (start > 0) {
+ const before = prevGrapheme(text, start);
+ const before_cat = hxCatAt(text, before);
+ if (!hxIsWs(before_cat) or before_cat == .eol) break;
+ start = before;
+ }
return .{ .anchor = start, .head = word_end };
}
@@ -1294,6 +1403,38 @@ test "hx find targets" {
try std.testing.expectEqual(@as(?usize, null), hxFindTarget(t, 0, 'z', true, false, 1));
}
+test "extended grapheme boundaries cover combining emoji flag and CJK text" {
+ const text = "a" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "🇧🇷" ++ "界";
+ const boundaries = [_]usize{ 0, 1, 4, 19, 27, 30 };
+ for (boundaries[0 .. boundaries.len - 1], boundaries[1..]) |start, end| {
+ try std.testing.expectEqual(end, nextGrapheme(text, start));
+ try std.testing.expectEqual(start, prevGrapheme(text, end));
+ }
+ // Stale byte offsets are repaired to a cluster boundary instead of being
+ // allowed to leak continuation bytes into cursor state.
+ try std.testing.expectEqual(@as(usize, 1), graphemeStart(text, 2));
+ try std.testing.expectEqual(@as(usize, 1), prevGrapheme(text, 3));
+ try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3));
+}
+
+test "Unicode find and word motion stay on grapheme boundaries" {
+ const lines = [_][]const u8{"\u{e9}x\u{e9}"};
+ try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'é', true, false, 1).?);
+
+ const text = "café 世界 ok\n";
+ const first = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 1, .next_word_start);
+ try std.testing.expectEqual(@as(usize, 0), first.anchor);
+ try std.testing.expectEqual(@as(usize, 6), first.head);
+ const second = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 2, .next_word_start);
+ try std.testing.expectEqual(@as(usize, 6), second.anchor);
+ try std.testing.expectEqual(@as(usize, 13), second.head);
+
+ // Long-word motions split on Unicode whitespace, not only ASCII spaces.
+ const nbsp = "alpha\u{a0}beta\n";
+ const long = hxWordMove(nbsp, .{ .anchor = 0, .head = 1 }, 1, .next_long_word_start);
+ try std.testing.expectEqual(@as(usize, 7), long.head);
+}
+
test "hx increment" {
const a = std.testing.allocator;
{
@@ -1335,7 +1476,7 @@ test "kindOf" {
test "char/line motions" {
const lines = [_][]const u8{ "alpha beta", " two words", "x" };
const c = Cursor{ .row = 0, .col = 5 };
- try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(c));
+ try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(&.{"hello"}, c));
try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, charRight(&lines, c));
try std.testing.expectEqual(Cursor{ .row = 1, .col = 5 }, lineDown(&lines, c));
try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, lineUp(&lines, Cursor{ .row = 1, .col = 5 }));
@@ -1440,6 +1581,19 @@ test "deleteChar" {
try std.testing.expectEqualStrings("abc", r2);
}
+test "Unicode edits replace and delete whole graphemes" {
+ const a = std.testing.allocator;
+ const content = "A" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "🇧🇷" ++ "界" ++ "Z";
+
+ const deleted = try deleteChar(a, content, .{ .row = 0, .col = 2 });
+ defer a.free(deleted);
+ try std.testing.expectEqualStrings("A👩🏽\u{200d}🚀🇧🇷界Z", deleted);
+
+ const replaced = try replaceChars(a, content, .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 19 }, '界');
+ defer a.free(replaced);
+ try std.testing.expectEqualStrings("A界界界界Z", replaced);
+}
+
test "deleteLines middle" {
const content = "one\ntwo\nthree\nfour";
const a = std.testing.allocator;
diff --git a/src/normal_input.zig b/src/normal_input.zig
index c6b2574f..4b047a7f 100644
--- a/src/normal_input.zig
+++ b/src/normal_input.zig
@@ -263,7 +263,7 @@ pub const Action = union(enum) {
goto: struct { target: Goto, count: u32, explicit_count: bool },
view: View,
find: struct { kind: Find, char: u21, count: u32 },
- replace_char: u8,
+ replace_char: u21,
match_bracket,
textobject: struct { char: u21, around: bool },
surround_add: u21,
@@ -409,8 +409,7 @@ pub fn parse(state: *State, key: Input) Result {
.replace => {
state.prefix = .none;
const char = key.literal() orelse return .ignored;
- if (char > 0x7f) return .ignored;
- return resultAction(.{ .replace_char = @intCast(char) });
+ return resultAction(.{ .replace_char = char });
},
.match => {
if (state.match_sub == .none) {
@@ -615,6 +614,13 @@ test "modified and special keys cannot satisfy literal continuations" {
try std.testing.expectEqual(State{}, state);
}
+test "replace accepts a Unicode literal" {
+ var state: State = .{};
+ try std.testing.expectEqual(Result.pending, parse(&state, input('r', &.{.prefix_replace})));
+ try std.testing.expectEqualDeep(Result{ .action = .{ .replace_char = '界' } }, parse(&state, input('界', &.{})));
+ try std.testing.expectEqual(State{}, state);
+}
+
test "once versus per-selection is semantic action metadata" {
try std.testing.expectEqual(Scope.once, (@as(Action, .search)).scope());
try std.testing.expectEqual(Scope.once, (Action{ .edit = .{ .kind = .undo, .count = 1 } }).scope());
diff --git a/src/pardes.zig b/src/pardes.zig
index 38e317c4..5bf03788 100644
--- a/src/pardes.zig
+++ b/src/pardes.zig
@@ -24,6 +24,7 @@ pub const animation = @import("animation.zig");
pub const panel_animation = @import("panel_animation.zig");
const ghostty_vt = @import("ghostty-vt");
const uucode = @import("uucode");
+const vaxis = @import("vaxis");
const mvzr = @import("mvzr");
const modal = @import("modal.zig");
const normal_input = @import("normal_input.zig");
@@ -1554,7 +1555,7 @@ const WordBounds = struct { lo: usize, hi: usize };
/// The whitespace-delimited word covering `col` in `str`. The topbar uses the
/// same bounds for pointer feedback and dispatch, so the word that lights up
/// is necessarily the word a middle click will execute.
-fn wordBoundsAtCol(str: []const u8, col: u16) ?WordBounds {
+fn wordBoundsAtCol(str: []const u8, col: usize) ?WordBounds {
if (col >= str.len or str[col] == ' ') return null;
var lo: usize = col;
while (lo > 0 and str[lo - 1] != ' ') lo -= 1;
@@ -1564,7 +1565,7 @@ fn wordBoundsAtCol(str: []const u8, col: u16) ?WordBounds {
}
/// the whitespace-delimited word covering `col` in `str` (topbar dispatch)
-fn wordAtCol(str: []const u8, col: u16) []const u8 {
+fn wordAtCol(str: []const u8, col: usize) []const u8 {
const bounds = wordBoundsAtCol(str, col) orelse return "";
return str[bounds.lo..bounds.hi];
}
@@ -2118,8 +2119,26 @@ pub const Surface = struct {
(std.unicode.utf8Decode(text[i .. i + n]) catch null)
else
null;
- var cp_slice = if (decoded == null) "\u{FFFD}" else text[i .. i + n];
- i += if (decoded == null) 1 else n;
+ var cp_slice: []const u8 = "\u{FFFD}";
+ var consumed: usize = 1;
+ if (decoded != null) {
+ // A surface cell is a grapheme, not a codepoint. Keeping the
+ // complete cluster makes combining marks visible and keeps ZWJ,
+ // modifier, flag and Indic sequences in the same screen cell
+ // that cursor/edit math treats as one unit.
+ var git = uucode.grapheme.utf8Iterator(text[i..]);
+ if (git.nextGrapheme()) |g| {
+ const candidate = text[i .. i + g.end];
+ if (std.unicode.utf8ValidateSlice(candidate)) {
+ cp_slice = candidate;
+ consumed = candidate.len;
+ } else {
+ cp_slice = text[i .. i + n];
+ consumed = n;
+ }
+ }
+ }
+ i += consumed;
var cp = decoded orelse 0xFFFD;
if (cp == '\r') continue;
// A Surface cell is already positioned, not a terminal byte
@@ -2137,8 +2156,10 @@ pub const Surface = struct {
cp = 0xFFFD;
cp_slice = "\u{FFFD}";
}
- const width: u16 = if (cp < 0x80) 1 else uucode.get(.width, cp);
- if (width == 0) continue;
+ const width: u16 = if (decoded == null or (cp < 0x80 and cp_slice.len == 1))
+ 1
+ else
+ @max(1, vaxis.gwidth.gwidth(cp_slice, .unicode));
// a DOUBLE-width glyph with one column left is not drawn at all.
// Writing it puts one cell in the surface and two on the glass, and
// when that column is the screen's last the terminal wraps the tail
@@ -2148,7 +2169,21 @@ pub const Surface = struct {
// problem. Reachable from any byte cut through wide text: hscroll's
// and soft wrap's both.
if (width == 2 and col + 1 >= end) break;
- s.set(col, y, cp_slice, style);
+ // The cross-host Cell ABI has seven payload bytes. Preserve a valid
+ // codepoint prefix when a modern emoji cluster is longer; its full
+ // width and edit boundary still come from the complete cluster.
+ var shown = cp_slice;
+ if (shown.len > @typeInfo(@FieldType(Cell, "text")).array.len) {
+ const cap = @typeInfo(@FieldType(Cell, "text")).array.len;
+ var prefix: usize = 0;
+ while (prefix < shown.len) {
+ const cp_len = std.unicode.utf8ByteSequenceLength(shown[prefix]) catch break;
+ if (prefix + cp_len > cap) break;
+ prefix += cp_len;
+ }
+ shown = if (prefix > 0) shown[0..prefix] else "\u{FFFD}";
+ }
+ s.set(col, y, shown, style);
if (width == 2 and col + 1 < end) {
// spacer: empty cell under the wide glyph's tail
s.set(col + 1, y, "", style);
@@ -2199,6 +2234,106 @@ test "surface print expands configured tabs and normalizes other controls" {
try std.testing.expectEqualStrings("A", cells[cell_count - 1].grapheme());
}
+test "surface print keeps combining and wide graphemes in their display cells" {
+ var cells: [8]Cell = @splat(.{});
+ var surface = Surface{ .cols = cells.len, .rows = 1, .cells = &cells };
+
+ const end = surface.print(0, 0, cells.len, "e\u{301}界👩🏽\u{200d}🚀1\u{fe0f}\u{20e3}A", .{});
+ try std.testing.expectEqual(@as(u16, 8), end);
+ try std.testing.expectEqualStrings("e\u{301}", cells[0].grapheme());
+ try std.testing.expectEqualStrings("界", cells[1].grapheme());
+ try std.testing.expectEqualStrings("", cells[2].grapheme());
+ try std.testing.expectEqualStrings("👩", cells[3].grapheme());
+ try std.testing.expectEqualStrings("", cells[4].grapheme());
+ try std.testing.expectEqualStrings("1\u{fe0f}\u{20e3}", cells[5].grapheme());
+ try std.testing.expectEqualStrings("", cells[6].grapheme());
+ try std.testing.expectEqualStrings("A", cells[7].grapheme());
+}
+
+test "insert and normal modes edit complete Unicode graphemes" {
+ const gpa = std.testing.allocator;
+ const p = try Pardes.init(gpa, .{ .cols = 60, .rows = 12 });
+ defer p.deinit();
+ const pane = try p.hxOpenFileContent("a" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "界" ++ "z");
+
+ pane.mode = .insert;
+ pane.cur_col = 22; // on z, after the CJK grapheme
+ p.insertKey(pane, .{ .cp = Key.left });
+ try std.testing.expectEqual(@as(i32, 19), pane.cur_col);
+ p.insertKey(pane, .{ .cp = Key.left });
+ try std.testing.expectEqual(@as(i32, 4), pane.cur_col);
+ p.insertKey(pane, .{ .cp = Key.right });
+ try std.testing.expectEqual(@as(i32, 19), pane.cur_col);
+ p.insertKey(pane, .{ .cp = Key.backspace });
+ try std.testing.expectEqualStrings("ae\u{301}界z", pane.file.?.content);
+ try std.testing.expectEqual(@as(i32, 4), pane.cur_col);
+
+ pane.mode = .normal;
+ pane.cur_col = 1;
+ p.normalDelete(pane, false);
+ try std.testing.expectEqualStrings("a界z", pane.file.?.content);
+ try std.testing.expectEqual(@as(i32, 1), pane.cur_col);
+ p.normalReplaceChar(pane, 'λ');
+ try std.testing.expectEqualStrings("aλz", pane.file.?.content);
+
+ file_pane.setContent(p, &pane.file.?, try gpa.dupe(u8, "界a"));
+ pane.cur_col = 3;
+ pane.vsel = .{ .active = true, .row = 0, .col = 0, .explicit = true };
+ p.normalReplaceChar(pane, 'λ');
+ try std.testing.expectEqualStrings("λλ", pane.file.?.content);
+ try std.testing.expectEqual(@as(i32, 2), pane.cur_col);
+ try std.testing.expectEqual(@as(i32, 0), pane.vsel.col);
+
+ file_pane.setContent(p, &pane.file.?, try gpa.dupe(u8, "界a"));
+ pane.cur_col = 0;
+ pane.vsel = .{ .active = true, .row = 0, .col = 3, .explicit = true };
+ p.normalReplaceChar(pane, 'λ');
+ try std.testing.expectEqualStrings("λλ", pane.file.?.content);
+ try std.testing.expectEqual(@as(i32, 0), pane.cur_col);
+ try std.testing.expectEqual(@as(i32, 2), pane.vsel.col);
+}
+
+test "Unicode display cells map back to body and tag byte cursors" {
+ const gpa = std.testing.allocator;
+ const p = try Pardes.init(gpa, .{ .cols = 60, .rows = 12 });
+ defer p.deinit();
+ const pane = try p.hxOpenFileContent("a界e\u{301}z\n");
+ const f = &pane.file.?;
+ p.gpa.free(f.path);
+ f.path = try p.gpa.dupe(u8, "界e\u{301}.txt");
+
+ var frame = std.heap.ArenaAllocator.init(gpa);
+ defer frame.deinit();
+ _ = try p.render(frame.allocator());
+ const rect = p.rects[0];
+ const text_x = rect.x + config.GUTTER + config.PREFIX_W;
+ const body_y = rect.y + BOX_H;
+ // The second screen cell is the trailing half of the wide CJK grapheme.
+ // Both halves map to its one byte boundary.
+ p.update(.{ .mouse = .{ .button = config.select_button, .kind = .press, .col = text_x + 2, .row = body_y } });
+ p.update(.{ .mouse = .{ .button = config.select_button, .kind = .release, .col = text_x + 2, .row = body_y } });
+ try std.testing.expectEqual(@as(i32, 1), pane.cur_col);
+
+ // A click in the wide path glyph likewise becomes a byte cursor at the
+ // grapheme start; arrow motion then advances by the full UTF-8 cluster.
+ p.enterTagEdit(pane, 1);
+ try std.testing.expectEqual(@as(u16, 0), pane.tag_col);
+ p.tagInsertKey(pane, .{ .cp = Key.right });
+ try std.testing.expectEqual(@as(u16, 3), pane.tag_col);
+ p.tagInsertKey(pane, .{ .cp = Key.right });
+ try std.testing.expectEqual(@as(u16, 6), pane.tag_col);
+
+ const before = try gpa.dupe(u8, pane.tagSlice());
+ defer gpa.free(before);
+ p.enterTagEdit(pane, -1);
+ const insertion = pane.tag_col;
+ p.tagInsertKey(pane, .{ .cp = 'λ', .text = "λ" });
+ try std.testing.expectEqual(insertion + 2, pane.tag_col);
+ p.tagInsertKey(pane, .{ .cp = Key.backspace });
+ try std.testing.expectEqual(insertion, pane.tag_col);
+ try std.testing.expectEqualStrings(before, pane.tagSlice());
+}
+
test "tabbed file aligns syntax cursor and mouse while preserving virtual columns" {
const gpa = std.testing.allocator;
const p = try Pardes.init(gpa, .{ .cols = 60, .rows = 12 });
@@ -2920,12 +3055,13 @@ pub const Pane = struct {
/// the body mode a tag edit hijacked (tags are always insert); terminals
/// restore it on exit so clicking the tag never changes the pane's mode
tag_mode: Mode = .normal,
- /// THE tag coordinate space: byte columns into the WHOLE rendered tag,
- /// prefix ++ tail (tagText). One space, no conversions — the renderer, the
- /// mouse and every motion speak it directly. The prefix is live chrome, so
- /// it is selectable/yankable/executable but READ-ONLY: every edit op
- /// measures from `edit0` (= tagPrefix().len, the first editable column) and
- /// does nothing left of it.
+ /// THE tag coordinate space: UTF-8 byte offsets into the WHOLE rendered
+ /// tag, prefix ++ tail (tagText), always on grapheme boundaries. Motions
+ /// and edits use these offsets; rendering and pointer input convert at the
+ /// screen boundary. The prefix is live chrome, so it is selectable,
+ /// yankable and executable but READ-ONLY: every edit op measures from
+ /// `edit0` (= tagPrefix().len, the first editable byte) and does nothing
+ /// left of it.
tag_col: u16 = 0,
tag_anchor: u16 = 0,
ed_undo: [term_pane.history_max]term_pane.Snapshot = undefined,
@@ -4540,7 +4676,7 @@ pub const Pardes = struct {
leader_on: bool = false,
leader_keys: [4]u8 = undefined,
leader_n: u8 = 0,
- /// the TOPBAR holds the keyboard, parked at this column of the rendered
+ /// the TOPBAR holds the keyboard, parked at this UTF-8 byte offset of the
/// row-0 line. Global like leader_on for the same reason: row 0 is not a
/// pane and never will be, so its one piece of focus state cannot live on
/// one. `null` = the panes have the keyboard, which is every other frame.
@@ -5401,12 +5537,9 @@ pub const Pardes = struct {
/// the row below kept its out at the pad. Now the column moves out
/// together, so the words stay in a line and a click walks down them.
///
- /// Bytes, not display columns — like every other tag coordinate (see
- /// Pane.tag_col). A multibyte path renders narrower than it measures, so
- /// the line is straight for ASCII paths and drifts by a column per wide
- /// character otherwise. That is the same approximation the plain
- /// right-align always made; fixing it is a job for the whole tag
- /// coordinate space, not for this scan.
+ /// Alignment is measured in display cells. Cursor and selection state use
+ /// UTF-8 byte offsets, then map through the same grapheme-width helpers at
+ /// the screen boundary.
///
/// Two rules make up the "when possible":
///
@@ -5450,7 +5583,7 @@ pub const Pardes = struct {
pane_tail;
const laid = if (q.tag_init) q.tagSlice() else words;
const lead = laid.len - std.mem.trimStart(u8, laid, " ").len;
- const q_end = (p.tagPrefix(q) catch continue).len + lead + std.mem.trimStart(u8, words, " ").len;
+ const q_end = file_pane.displayWidth(p.tagPrefix(q) catch continue) + lead + file_pane.displayWidth(std.mem.trimStart(u8, words, " "));
end = @max(end, @min(q_end, tw));
};
return end -| used;
@@ -5462,7 +5595,7 @@ pub const Pardes = struct {
fn tagText(p: *Pardes, arena: std.mem.Allocator, pane: *Pane) ![]u8 {
const prefix = try p.tagPrefix(pane);
const tail = curTail(pane);
- const gap = p.tagGap(pane, prefix.len + tail.len);
+ const gap = p.tagGap(pane, file_pane.displayWidth(prefix) + file_pane.displayWidth(tail));
const out = try arena.alloc(u8, prefix.len + gap + tail.len);
@memcpy(out[0..prefix.len], prefix);
@memset(out[prefix.len..][0..gap], ' ');
@@ -5476,8 +5609,8 @@ pub const Pardes = struct {
fn seedTail(p: *Pardes, pane: *Pane) void {
if (pane.tag_init) return;
const tail = curTail(pane);
- const prefix_len = (p.tagPrefix(pane) catch return).len;
- const gap = p.tagGap(pane, prefix_len + tail.len);
+ const prefix = (p.tagPrefix(pane) catch return);
+ const gap = p.tagGap(pane, file_pane.displayWidth(prefix) + file_pane.displayWidth(tail));
if (gap + tail.len > pane.tag_tail.len) return;
@memset(pane.tag_tail[0..gap], ' ');
@memcpy(pane.tag_tail[gap..][0..tail.len], tail);
@@ -5486,7 +5619,7 @@ pub const Pardes = struct {
}
/// focus the tag for editing, seeding the tail on first touch and parking
- /// the cursor at `col` — a column of the RENDERED tag (see tag_col).
+ /// the byte cursor at the grapheme displayed under screen column `col`.
/// NEGATIVE means the tail's first WORD, which is where `:` and a tagline
/// hop land: a place no click can name, so it needs no sentinel of its own
/// and the callers need no prefix length.
@@ -5509,7 +5642,12 @@ pub const Pardes = struct {
// the door. The spaces stay editable; h and Left still walk into them.
const tail = pane.tagSlice();
const lead: i32 = @intCast(tail.len - std.mem.trimStart(u8, tail, " ").len);
- pane.tag_col = @intCast(if (col < 0) @min(edit0 + lead, end) else std.math.clamp(col, 0, end));
+ if (col < 0) {
+ pane.tag_col = @intCast(@min(edit0 + lead, end));
+ } else {
+ const text = p.tagText(p.scratch.allocator(), pane) catch return;
+ pane.tag_col = @intCast(@min(text.len, file_pane.rawAtDisplay(text, @intCast(col))));
+ }
}
fn exitTagEdit(pane: *Pane) void {
@@ -5538,7 +5676,7 @@ pub const Pardes = struct {
const text = p.tagText(p.scratch.allocator(), pane) catch return null;
if (pane.tag_sel) {
const b = tagSelBounds(pane);
- const hi = @min(b.hi + 1, text.len);
+ const hi = modal.nextGrapheme(text, b.hi);
return if (hi > b.lo) text[b.lo..hi] else null;
}
const b = config.wordBounds(text, @min(@as(usize, pane.tag_col), text.len));
@@ -5580,14 +5718,19 @@ pub const Pardes = struct {
}
switch (key.cp) {
Key.backspace => if (pane.tag_col > edit0) {
- pane.removeTagByte(pane.tag_col - edit0 - 1);
- pane.tag_col -= 1;
+ const text = p.tagText(p.scratch.allocator(), pane) catch return;
+ const prev = @max(@as(usize, edit0), modal.prevGrapheme(text, pane.tag_col));
+ const count = @as(usize, pane.tag_col) - prev;
+ for (0..count) |_| pane.removeTagByte(prev - edit0);
+ pane.tag_col = @intCast(prev);
},
Key.left => if (pane.tag_col > 0) {
- pane.tag_col -= 1;
+ const text = p.tagText(p.scratch.allocator(), pane) catch return;
+ pane.tag_col = @intCast(modal.prevGrapheme(text, pane.tag_col));
},
Key.right => if (pane.tag_col < end) {
- pane.tag_col += 1;
+ const text = p.tagText(p.scratch.allocator(), pane) catch return;
+ pane.tag_col = @intCast(modal.nextGrapheme(text, pane.tag_col));
},
Key.home => pane.tag_col = 0,
Key.end => pane.tag_col = end,
@@ -5601,8 +5744,8 @@ pub const Pardes = struct {
const lo = @min(r.anchor, r.head);
const hi = @max(r.anchor, r.head);
pane.tag_col = @intCast(modal.hxCursor(text, r));
- pane.tag_sel = hi > lo + 1; // a 1-wide range IS the block cursor
- if (pane.tag_sel) pane.tag_anchor = @intCast(if (r.head > r.anchor) lo else hi - 1);
+ pane.tag_sel = modal.nextGrapheme(text, lo) < hi; // one grapheme IS the block cursor
+ if (pane.tag_sel) pane.tag_anchor = @intCast(if (r.head > r.anchor) lo else modal.prevGrapheme(text, hi));
}
/// normal mode ON the tag — where `:` lands. The body's own helix motions
@@ -5757,7 +5900,7 @@ pub const Pardes = struct {
// on it does. Leave the bar FIRST — `Kill` lives up here and tears the
// session down, the same hazard the pane-tag chord has with `Del`.
if (hit(key, config.look_key) or hit(key, config.exec_key)) {
- const word = wordAtCol(bar, @intCast(cur));
+ const word = wordAtCol(bar, cur);
p.topbar_col = null;
if (word.len > 0) _ = p.execute(p.active, word);
return;
@@ -5819,7 +5962,7 @@ pub const Pardes = struct {
while (count_it.next()) |line| : (count_row += 1) {
if (count_row < r0 or count_row > r1) continue;
const b0 = @min(file_pane.renderedLineByteCol(pane, count_row, line, c0), line.len);
- const b1 = @min(file_pane.renderedLineByteCol(pane, count_row, line, c1) + 1, line.len);
+ const b1 = modal.nextGrapheme(line, @min(file_pane.renderedLineByteCol(pane, count_row, line, c1), line.len));
total += b1 - b0 + @intFromBool(selected > 0);
selected += 1;
}
@@ -5836,7 +5979,7 @@ pub const Pardes = struct {
}
first = false;
const b0 = @min(file_pane.renderedLineByteCol(pane, v, line, c0), line.len);
- const b1 = @min(file_pane.renderedLineByteCol(pane, v, line, c1) + 1, line.len);
+ const b1 = modal.nextGrapheme(line, @min(file_pane.renderedLineByteCol(pane, v, line, c1), line.len));
@memcpy(out[at..][0 .. b1 - b0], line[b0..b1]);
at += b1 - b0;
}
@@ -5891,10 +6034,17 @@ pub const Pardes = struct {
/// the word under the modal cursor as a pane-local selection (paneText
/// coords: row 0 is the tag; file panes carry the line-number prefix)
- fn cursorWordSel(pane: *Pane) Sel {
+ fn cursorWordSel(p: *Pardes, pane: *Pane) Sel {
const w = pane.wrapRow(pane.cur_row, pane.cur_col);
const vrow = w.row + @as(i32, BOX_H);
- const vcol = if (pane.file != null) file_pane.displayOffset(pane, pane.cur_row, w.at, pane.cur_col) + @as(i32, config.PREFIX_W) else pane.cur_col;
+ const vcol = if (pane.file != null)
+ file_pane.displayOffset(pane, pane.cur_row, w.at, pane.cur_col) + @as(i32, config.PREFIX_W)
+ else blk: {
+ const pl = p.paneCursorLines(pane) catch break :blk pane.cur_col;
+ const local = pane.cur_row - pl.row0;
+ if (local < 0 or @as(usize, @intCast(local)) >= pl.lines.len) break :blk pane.cur_col;
+ break :blk file_pane.lineDisplayOffset(pl.lines[@intCast(local)], @intCast(@max(0, w.at)), @intCast(@max(0, pane.cur_col)));
+ };
return .{ .state = .done, .c0 = vcol, .c1 = vcol, .r0 = vrow, .r1 = vrow };
}
@@ -5930,8 +6080,13 @@ pub const Pardes = struct {
const visible = clicked.r0 - @as(i32, BOX_H);
const wrapped = pane.wrapAt(visible);
const row = wrapped.line;
- const col = if (pane.file != null)
- file_pane.byteAtRowDisplay(pane, wrapped.line, wrapped.at, clicked.c0 - @as(i32, config.PREFIX_W))
+ const col = if (clicked.r0 >= BOX_H)
+ p.paneByteAtDisplay(
+ pane,
+ wrapped.line,
+ wrapped.at,
+ clicked.c0 - (if (pane.file != null) @as(i32, config.PREFIX_W) else 0),
+ )
else
clicked.c0;
var result: PointerOperand = .{ .row = row, .col = col };
@@ -6157,7 +6312,7 @@ pub const Pardes = struct {
// to the file-ish word under the cursor.
if (pane.mode == .normal and (hit(key, config.look_key) or hit(key, config.exec_key))) {
const cmd = if (hit(key, config.look_key)) config.look_cmd else config.exec_cmd;
- pane.pinCursor();
+ p.pinPaneCursor(pane);
const explicit = (p.native_images and hasPdfSelection(pane)) or
(pane.vsel.active and pane.vsel.explicit) or pane.msel.active;
if (explicit) {
@@ -6169,7 +6324,7 @@ pub const Pardes = struct {
return;
}
}
- const sel = p.expandedSel(pane, cursorWordSel(pane)) orelse return;
+ const sel = p.expandedSel(pane, p.cursorWordSel(pane)) orelse return;
const word = p.selectionText(pane, sel) catch return;
p.runBuiltin(cmd, p.active, "", word);
return;
@@ -6439,6 +6594,29 @@ pub const Pardes = struct {
return .{ .lines = try term_pane.cursorLines(p, pane), .row0 = 0 };
}
+ fn paneByteAtDisplay(p: *Pardes, pane: *Pane, row: i32, from_raw: i32, display_col: i32) i32 {
+ if (pane.file != null)
+ return file_pane.byteAtRowDisplay(pane, row, from_raw, display_col);
+ const pl = p.paneCursorLines(pane) catch return @max(0, from_raw + display_col);
+ const local = row - pl.row0;
+ if (local < 0 or @as(usize, @intCast(local)) >= pl.lines.len)
+ return @max(0, from_raw + display_col);
+ return @intCast(file_pane.byteAtDisplayFrom(
+ pl.lines[@intCast(local)],
+ @intCast(@max(0, from_raw)),
+ @intCast(@max(0, display_col)),
+ ));
+ }
+
+ /// Freeze the live terminal cursor into the modal coordinate space. The
+ /// emulator reports screen cells; editing state stores UTF-8 byte offsets.
+ fn pinPaneCursor(p: *Pardes, pane: *Pane) void {
+ if (pane.cur_pinned) return;
+ pane.pinCursor();
+ if (pane.file == null and !hasPdf(pane))
+ pane.cur_col = p.paneByteAtDisplay(pane, pane.cur_row, 0, pane.cur_col);
+ }
+
fn toModalCursor(pane: *Pane, pl: PaneLines) modal.Cursor {
const r: i32 = pane.cur_row - pl.row0;
return .{ .row = @intCast(@max(0, r)), .col = @intCast(@max(0, pane.cur_col)) };
@@ -6450,6 +6628,18 @@ pub const Pardes = struct {
pane.cur_pinned = true;
}
+ fn insertVerticalCursor(lines: []const []const u8, c: modal.Cursor, down: bool) modal.Cursor {
+ if (lines.len == 0) return c;
+ const row = if (down) @min(c.row + 1, lines.len - 1) else c.row -| 1;
+ const target = lines[row];
+ if (target.len == 0) return .{ .row = row, .col = 0 };
+ const source = if (c.row < lines.len) lines[c.row] else "";
+ const goal = file_pane.rawDisplayCol(source, c.col);
+ const mapped = file_pane.rawAtDisplay(target, goal);
+ const last = modal.prevGrapheme(target, target.len);
+ return .{ .row = row, .col = modal.graphemeStart(target, @min(mapped, last)) };
+ }
+
// ---- helix range plumbing (see modal.zig "helix range engine") ----
// The pane's cursor + vsel cells render ONE helix gap range over the flat
// motion surface. Every motion builds the current range, transforms it the
@@ -7222,9 +7412,8 @@ pub const Pardes = struct {
/// f/t/F/T: anchor at the old cursor cell, head on the hit (not found: no move)
fn findMove(pane: *Pane, pl: PaneLines, text: []const u8, range: modal.HxRange, ch: u21, fwd: bool, till: bool, cnt: usize) void {
- if (ch > 0x7f) return; // ponytail: ASCII targets only (byte columns)
const cur = modal.hxCursor(text, range);
- const t = modal.hxFindTarget(text, cur, @intCast(ch), fwd, till, cnt) orelse return;
+ const t = modal.hxFindTarget(text, cur, ch, fwd, till, cnt) orelse return;
const res = if (pane.select)
modal.hxPutCursor(text, range, t, true)
else
@@ -7236,13 +7425,17 @@ pub const Pardes = struct {
fn verticalMove(pane: *Pane, pl: PaneLines, text: []const u8, range: modal.HxRange, down: bool, cnt: usize) void {
const cur = modal.hxCursor(text, range);
const pos = panePos(pane, text, cur);
- const goal: usize = if (pane.sticky_col >= 0) @intCast(pane.sticky_col) else pos.col;
+ const goal: usize = if (pane.sticky_col >= 0)
+ @intCast(pane.sticky_col)
+ else
+ file_pane.rawDisplayCol(modal.lineSlice(text, pos.row), pos.col);
// modal.hxVertTarget with the row we already have and the indexed
// offset conversion — it would otherwise recount the buffer's newlines
// and walk to the target line, two more full passes per j/k
const last_row = paneLineCount(pane, text) - 1;
const nline = if (down) @min(pos.row + @max(1, cnt), last_row) else pos.row -| @max(1, cnt);
- const t = paneOff(pane, text, .{ .row = nline, .col = goal });
+ const target_col = file_pane.rawAtDisplay(modal.lineSlice(text, nline), goal);
+ const t = paneOff(pane, text, .{ .row = nline, .col = target_col });
// extend mode never walks onto the empty trailing line (helix)
if (pane.select and t == text.len and text.len > 0 and text[text.len - 1] == '\n') return;
setPaneRange(pane, pl, text, modal.hxPutCursor(text, range, t, pane.select), false);
@@ -7373,7 +7566,7 @@ pub const Pardes = struct {
/// complete before this function runs; this switch reads document state
/// only to execute the already-recognized action.
fn executeNormalAction(p: *Pardes, pane: *Pane, semantic: normal_input.Action) void {
- pane.pinCursor();
+ p.pinPaneCursor(pane);
const pl = p.paneCursorLines(pane) catch return;
const text = p.flatSurface(pane, pl) catch return;
const lines = pl.lines;
@@ -7410,7 +7603,8 @@ pub const Pardes = struct {
.column => {
const line = modal.hxLineOf(text, cur);
const ls = modal.lineStartOffset(text, line);
- return pointMove(pane, pl, text, range, @min(ls + (@as(usize, go.count) - 1), modal.hxLineEndIdx(text, line)));
+ const slice = text[ls..modal.hxLineEndIdx(text, line)];
+ return pointMove(pane, pl, text, range, ls + modal.graphemeAtColumn(slice, @as(usize, go.count) - 1));
},
.view_top => return gotoWindow(pane, pl, text, range, .top, go.count),
.view_center => return gotoWindow(pane, pl, text, range, .center, go.count),
@@ -7506,14 +7700,14 @@ pub const Pardes = struct {
.next_long_word_end => return wordMove(pane, pl, text, range, move.count, .next_long_word_end),
},
.repeat_find => |count| {
- if (pane.find_op == 0 or pane.find_ch > 0x7f) return;
+ if (pane.find_op == 0) return;
const fwd = pane.find_op == config.find_char_fwd or pane.find_op == config.till_char_fwd;
const till = pane.find_op == config.till_char_fwd or pane.find_op == config.till_char_back;
var repeated = range;
var moved = false;
for (0..count) |_| {
const cc = modal.hxCursor(text, repeated);
- const target = modal.hxFindTarget(text, cc, @intCast(pane.find_ch), fwd, till, 1) orelse break;
+ const target = modal.hxFindTarget(text, cc, pane.find_ch, fwd, till, 1) orelse break;
repeated = if (pane.select)
modal.hxPutCursor(text, repeated, target, true)
else
@@ -7815,7 +8009,7 @@ pub const Pardes = struct {
pane.tag_sel = false;
pane.mode = .insert;
pane.pending = 0;
- // tag_col is a rendered-tag column; the prompt offset slices tag_tail.
+ // tag_col and the prompt offset are both UTF-8 byte offsets.
pane.tag_col = @intCast((p.tagPrefix(pane) catch return).len + pane.tag_tail_len);
}
@@ -8531,7 +8725,7 @@ pub const Pardes = struct {
/// cursor on its FIRST — the same shape the old terminal stepper left, and
/// the reason `col0` is the position the walk compares against.
fn landLookSpot(p: *Pardes, id: usize, pane: *Pane, spot: LookSpot) void {
- pane.pinCursor(); // fresh out of tty mode the cursor still tracks the shell
+ p.pinPaneCursor(pane); // fresh out of tty mode the cursor still tracks the shell
pane.vsel = .{ .active = true, .row = spot.row, .col = spot.col1, .explicit = true };
pane.msel.active = false;
pane.nsel = 0;
@@ -8572,7 +8766,7 @@ pub const Pardes = struct {
fn enterInsert(p: *Pardes, pane: *Pane, where: InsertAt, cnt: usize) void {
if (hasPdf(pane)) return;
- pane.pinCursor();
+ p.pinPaneCursor(pane);
// snapshot once per insert session (WITH the pre-insert selection) so
// `u` undoes the whole session and restores what was selected
p.pushUndo(pane);
@@ -8694,7 +8888,7 @@ pub const Pardes = struct {
// its only pardes use (the acme chords) needs explicit selections
// anyway, and those never enter insert mode
pane.vsel.active = false;
- if (!pane.cur_pinned) pane.pinCursor();
+ if (!pane.cur_pinned) p.pinPaneCursor(pane);
// arrows and paging are pure motion over the WHOLE surface, so they
// run before editText — a terminal must not freeze shell rows into an
// edit buffer just because you walked across them
@@ -8703,10 +8897,10 @@ pub const Pardes = struct {
const pl = p.paneCursorLines(pane) catch return;
const cur0 = toModalCursor(pane, pl);
const nc = switch (key.cp) {
- Key.left => modal.charLeft(cur0),
+ Key.left => modal.charLeft(pl.lines, cur0),
Key.right => modal.charRight(pl.lines, cur0),
- Key.up => modal.lineUp(pl.lines, cur0),
- Key.down => modal.lineDown(pl.lines, cur0),
+ Key.up => insertVerticalCursor(pl.lines, cur0, false),
+ Key.down => insertVerticalCursor(pl.lines, cur0, true),
else => cur0,
};
fromModalCursor(pane, pl, nc);
@@ -8829,9 +9023,11 @@ pub const Pardes = struct {
},
Key.backspace => {
if (pane.cur_col > 0) {
- const new = modal.deleteChar(p.gpa, text, .{ .row = c.row, .col = c.col - 1 }) catch return;
+ const line = modal.lineSlice(text, c.row);
+ const prev = modal.prevGrapheme(line, c.col);
+ const new = modal.deleteChar(p.gpa, text, .{ .row = c.row, .col = prev }) catch return;
p.setEditText(pane, new);
- pane.cur_col -= 1;
+ pane.cur_col = @intCast(prev);
pane.cur_pinned = true;
pane.ensureCursorVisible();
return;
@@ -9115,17 +9311,19 @@ pub const Pardes = struct {
.{ .row = @intCast(@max(0, b.lo_row - row0)), .col = @intCast(@max(0, b.lo_col)) }
else blk: {
const hrow: usize = @intCast(@max(0, b.hi_row - row0));
- const gcol = @as(usize, @intCast(@max(0, b.hi_col))) + 1;
- if (gcol > modal.lineSlice(eb.text, hrow).len) break :blk .{ .row = hrow + 1, .col = 0 };
+ const hi_col: usize = @intCast(@max(0, b.hi_col));
+ const high_line = modal.lineSlice(eb.text, hrow);
+ if (hi_col >= high_line.len) break :blk .{ .row = hrow + 1, .col = 0 };
+ const gcol = modal.nextGrapheme(high_line, hi_col);
break :blk .{ .row = hrow, .col = gcol };
};
const out = modal.insertAt(p.gpa, eb.text, at, y) catch return;
p.setEditText(pane, out);
- pane.vsel = .{ .active = y.len > 1, .row = @as(i32, @intCast(at.row)) + row0, .col = @intCast(at.col), .explicit = false };
+ pane.vsel = .{ .active = modal.nextGrapheme(y, 0) < y.len, .row = @as(i32, @intCast(at.row)) + row0, .col = @intCast(at.col), .explicit = false };
const end = modal.advanceBy(at, y);
if (end.col > 0) {
pane.cur_row = @as(i32, @intCast(end.row)) + row0;
- pane.cur_col = @intCast(end.col - 1);
+ pane.cur_col = @intCast(modal.prevGrapheme(modal.lineSlice(out, end.row), end.col));
} else {
pane.cur_row = @as(i32, @intCast(end.row -| 1)) + row0;
pane.cur_col = @intCast(modal.lineSlice(out, end.row -| 1).len);
@@ -9261,7 +9459,7 @@ pub const Pardes = struct {
const r0: usize = @intCast(@max(0, @min(pane.msel.r0, pane.msel.r1) - row0));
const r1: usize = @intCast(@max(0, @max(pane.msel.r0, pane.msel.r1) - row0));
const llen = modal.lineSlice(content, r1).len;
- return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = llen -| 1 } };
+ return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else modal.prevGrapheme(modal.lineSlice(content, r1), llen) } };
}
const c = modal.Cursor{ .row = @intCast(@max(0, pane.cur_row - row0)), .col = @intCast(@max(0, pane.cur_col)) };
return .{ .a = c, .b = c };
@@ -9281,13 +9479,27 @@ pub const Pardes = struct {
/// `r<ch>`: overwrite the selection (or the cursor char) with ch —
/// newlines included (helix), so `xrz` joins the selected lines
- fn normalReplaceChar(p: *Pardes, pane: *Pane, ch: u8) void {
+ fn normalReplaceChar(p: *Pardes, pane: *Pane, ch: u21) void {
pane.select = false;
const eb = p.editTextEol(pane, selRows(pane)) orelse return;
+ const before = paneRange(pane, eb.text, eb.row0);
+ const lo = @min(before.anchor, before.head);
+ const hi = @max(before.anchor, before.head);
+ var graphemes: usize = 0;
+ var at = lo;
+ while (at < hi) : (graphemes += 1) at = modal.nextGrapheme(eb.text, at);
+ var encoded: [4]u8 = undefined;
+ const encoded_len = std.unicode.utf8Encode(ch, &encoded) catch return;
const r = selRange(pane, eb.text, eb.row0);
p.pushUndo(pane);
const new = modal.replaceChars(p.gpa, eb.text, r.a, r.b, ch) catch return;
p.setEditText(pane, new);
+ const end = lo + graphemes * encoded_len;
+ const mapped: modal.HxRange = if (before.head < before.anchor)
+ .{ .anchor = end, .head = lo }
+ else
+ .{ .anchor = lo, .head = end };
+ setPaneRange(pane, .{ .lines = &.{}, .row0 = eb.row0 }, new, mapped, pane.vsel.explicit);
}
/// `R`: replace the selection (or the cursor char) with the DEFAULT
@@ -9307,11 +9519,11 @@ pub const Pardes = struct {
const new = modal.replaceRange(p.gpa, eb.text, r.a, r.b, y) catch return;
p.setEditText(pane, new);
pane.msel.active = false;
- pane.vsel = .{ .active = y.len > 1, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false };
+ pane.vsel = .{ .active = modal.nextGrapheme(y, 0) < y.len, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false };
const end = modal.advanceBy(r.a, y);
if (end.col > 0) {
pane.cur_row = @as(i32, @intCast(end.row)) + eb.row0;
- pane.cur_col = @intCast(end.col - 1);
+ pane.cur_col = @intCast(modal.prevGrapheme(modal.lineSlice(new, end.row), end.col));
} else {
// the yank ended in '\n': the cursor lands ON that newline
pane.cur_row = @as(i32, @intCast(end.row -| 1)) + eb.row0;
@@ -9637,8 +9849,8 @@ pub const Pardes = struct {
const new = modal.replaceRange(p.gpa, eb.text, r.a, r.b, rep) catch return;
p.setEditText(pane, new);
const start = modal.lineStartOffset(new, r.a.row) + r.a.col;
- const cc = modal.hxPos(new, start + rep.len - 1);
- pane.vsel = .{ .active = rep.len > 1, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false };
+ const cc = modal.hxPos(new, modal.prevGrapheme(new, start + rep.len));
+ pane.vsel = .{ .active = modal.nextGrapheme(rep, 0) < rep.len, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false };
pane.msel.active = false;
pane.cur_row = @as(i32, @intCast(cc.row)) + eb.row0;
pane.cur_col = @intCast(cc.col);
@@ -9728,14 +9940,15 @@ pub const Pardes = struct {
const eb = p.editTextEol(pane, selRows(pane)) orelse return;
const r = selRange(pane, eb.text, eb.row0);
p.pushUndo(pane);
- var new = modal.insertAt(p.gpa, eb.text, .{ .row = r.b.row, .col = r.b.col + 1 }, &[1]u8{pr.c}) catch return;
+ const close_col = modal.nextGrapheme(modal.lineSlice(eb.text, r.b.row), r.b.col);
+ var new = modal.insertAt(p.gpa, eb.text, .{ .row = r.b.row, .col = close_col }, &[1]u8{pr.c}) catch return;
p.setEditText(pane, new);
new = modal.insertAt(p.gpa, new, .{ .row = r.a.row, .col = r.a.col }, &[1]u8{pr.o}) catch return;
p.setEditText(pane, new);
pane.msel.active = false;
pane.vsel = .{ .active = true, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false };
pane.cur_row = @as(i32, @intCast(r.b.row)) + eb.row0;
- pane.cur_col = @intCast(r.b.col + 1 + @as(usize, if (r.a.row == r.b.row) 1 else 0));
+ pane.cur_col = @intCast(close_col + @as(usize, if (r.a.row == r.b.row) 1 else 0));
pane.cur_pinned = true;
pane.sticky_col = -1;
pane.ensureCursorVisible();
@@ -10329,7 +10542,13 @@ pub const Pardes = struct {
// next cursor move or left wheel pulls it back
} else if (pane.file != null and !p.settings.wrap) {
// wrapped there is nothing off to the right to reach
- pane.hscroll = @max(0, pane.hscroll + (if (m.button == .wheel_right) config.wheel_cols else -config.wheel_cols));
+ const line = file_pane.sourceLine(pane, pane.cur_row);
+ const visual = file_pane.rawDisplayCol(line, @intCast(@max(0, pane.hscroll)));
+ const next: usize = if (m.button == .wheel_right)
+ visual +| @as(usize, @intCast(config.wheel_cols))
+ else
+ visual -| @as(usize, @intCast(config.wheel_cols));
+ pane.hscroll = @intCast(file_pane.rawAtDisplay(line, next));
}
},
config.select_button => switch (m.kind) {
@@ -10474,7 +10693,8 @@ pub const Pardes = struct {
// click (a stray select click must never Kill)
if (m.button == config.exec_button) {
var tb_buf: [1200]u8 = undefined;
- const word = wordAtCol(p.topbar(&tb_buf), mcol);
+ const bar = p.topbar(&tb_buf);
+ const word = wordAtCol(bar, file_pane.rawAtDisplay(bar, mcol));
// a topbar word runs on the PRESS — there is no
// drag to chord into, so the argument is simply
// whatever is selected right now: select a word,
@@ -10615,9 +10835,9 @@ pub const Pardes = struct {
// row is ignored (a tag is one line) and the anchor stays where
// the press put it, so this is the mouse's `v`
const pane = p.panes[d.id] orelse return;
- const end: i32 = @intCast((p.tagPrefix(pane) catch return).len + pane.tag_tail_len);
const c = @as(i32, mcol) - @as(i32, p.rects[d.id].x + config.GUTTER);
- pane.tag_col = @intCast(std.math.clamp(c, 0, end));
+ const text = p.tagText(p.scratch.allocator(), pane) catch return;
+ pane.tag_col = @intCast(@min(text.len, file_pane.rawAtDisplay(text, @intCast(@max(0, c)))));
pane.tag_sel = pane.tag_col != pane.tag_anchor;
},
.none => {},
@@ -10722,10 +10942,12 @@ pub const Pardes = struct {
// pointer whether or not that row is a continuation
const w = pane.wrapAt(body_vis);
pane.cur_row = w.line;
- pane.cur_col = if (pane.file != null)
- file_pane.byteAtRowDisplay(pane, w.line, w.at, sl.c1 - @as(i32, config.PREFIX_W))
- else
- sl.c1;
+ pane.cur_col = p.paneByteAtDisplay(
+ pane,
+ w.line,
+ w.at,
+ sl.c1 - (if (pane.file != null) @as(i32, config.PREFIX_W) else 0),
+ );
pane.cur_pinned = true;
if (!pane.isTerminal()) pane.mode = .normal;
pane.msel.active = false;
@@ -10874,8 +11096,8 @@ pub const Pardes = struct {
const w1 = pane.wrapAt(@max(0, sl.r1 - @as(i32, BOX_H)));
const row0 = w0.line;
const row1 = w1.line;
- const col0 = @max(0, sl.c0 - pfx) + w0.at;
- const col1 = @max(0, sl.c1 - pfx) + w1.at;
+ const col0 = p.paneByteAtDisplay(pane, w0.line, w0.at, sl.c0 - pfx);
+ const col1 = p.paneByteAtDisplay(pane, w1.line, w1.at, sl.c1 - pfx);
pane.cur_row = row1;
pane.cur_col = col1;
pane.cur_pinned = true;
@@ -10898,11 +11120,11 @@ pub const Pardes = struct {
p.pushUndo(pane);
const new = modal.insertAt(p.gpa, f.content, at, y) catch return;
file_pane.setContent(p, f, new);
- pane.vsel = .{ .active = y.len > 1, .row = @intCast(at.row), .col = @intCast(at.col), .explicit = false };
+ pane.vsel = .{ .active = modal.nextGrapheme(y, 0) < y.len, .row = @intCast(at.row), .col = @intCast(at.col), .explicit = false };
const end = modal.advanceBy(at, y);
if (end.col > 0) {
pane.cur_row = @intCast(end.row);
- pane.cur_col = @intCast(end.col - 1);
+ pane.cur_col = @intCast(modal.prevGrapheme(modal.lineSlice(f.content, end.row), end.col));
} else {
// the register ended in '\n': the cursor lands ON that newline
pane.cur_row = @intCast(end.row -| 1);
@@ -11192,6 +11414,14 @@ pub const Pardes = struct {
if (comptime !pdf_enabled) return;
if (pane.pdf == null) return;
var state = paneNormalState(pane);
+ // A PDF has no body character to find, so its bare `f` is the direct
+ // document-outline door. Prefix continuations still go through the
+ // shared parser, and text/terminal panes retain `f<char>` unchanged.
+ if (state.prefix == .none and isPrefix(key, 'f')) {
+ state.clear();
+ putPaneNormalState(pane, state);
+ return p.runBuiltin(.PdfSections, p.active, "", null);
+ }
const parsed = normal_input.parse(&state, normalInput(key));
putPaneNormalState(pane, state);
switch (parsed) {
@@ -12584,17 +12814,20 @@ pub const Pardes = struct {
const w: u16 = @intCast(iw);
if (w < tw) _ = s.print(tx + tw - w, row, w, ibuf[0..iw], msg_style);
}
- // ...and the cursor follows the text it edits. tag_col is a
- // rendered-tag column, so the prompt's own column is it minus where
- // the marker starts; a cursor LEFT of that is still over the part
+ // ...and the cursor follows the text it edits. tag_col is a byte
+ // offset, so the prompt maps its suffix through display widths; a
+ // cursor LEFT of the marker is still over the part
// of the tag that stayed on the tagline, and the tag cursor
// renderPane already placed there is the right one.
if (id != p.active) continue;
const at = prompt_at orelse continue;
const prompt0 = (p.tagPrefix(pane) catch continue).len + at;
const col = @as(usize, pane.tag_col);
- if (col >= prompt0 and col - prompt0 < tw)
- s.cursor = .{ .x = tx + @as(u16, @intCast(col - prompt0)), .y = row, .bar = pane.mode == .insert };
+ if (col >= prompt0) {
+ const prompt_col = file_pane.displayWidth(text[0..@min(col - prompt0, text.len)]);
+ if (prompt_col < tw)
+ s.cursor = .{ .x = tx + @as(u16, @intCast(prompt_col)), .y = row, .bar = pane.mode == .insert };
+ }
}
// global tagbar: full width, top row
@@ -12617,9 +12850,11 @@ pub const Pardes = struct {
// feedback at all. Paint the same word the click dispatcher resolves,
// immediately, while leaving whitespace inert.
if (p.pointer_inside and p.hover_row < TOPBAR_H) {
- if (wordBoundsAtCol(p.topbar(&tb_buf), p.hover_col)) |bounds| {
- var col: usize = bounds.lo;
- while (col < bounds.hi and col < s.cols) : (col += 1) {
+ const bar = p.topbar(&tb_buf);
+ if (wordBoundsAtCol(bar, file_pane.rawAtDisplay(bar, p.hover_col))) |bounds| {
+ var col = file_pane.rawDisplayCol(bar, bounds.lo);
+ const hi = file_pane.rawDisplayCol(bar, bounds.hi);
+ while (col < hi and col < s.cols) : (col += 1) {
const cell = s.at(@intCast(col), 0);
cell.default = false;
cell.style.bg = .{ .rgb = th.sel_bg };
@@ -12630,9 +12865,10 @@ pub const Pardes = struct {
// the topbar's cursor, if it has the keyboard. AFTER the pane loop on
// purpose: there is exactly one Surface cursor and the bar's must beat
// the active pane's. Always a block — the bar has no insert mode.
- if (p.topbar_col) |c| if (c < s.cols) {
- s.cursor = .{ .x = c, .y = 0, .bar = false };
- };
+ if (p.topbar_col) |c| {
+ const col = file_pane.rawDisplayCol(p.topbar(&tb_buf), c);
+ if (col < s.cols) s.cursor = .{ .x = @intCast(col), .y = 0, .bar = false };
+ }
// resize-handle hint / drag previews: a dash overlay that keeps the
// underlying colors (border drags + hover), or the move indicator. A
@@ -13046,7 +13282,7 @@ pub const Pardes = struct {
};
}
// tag char selection highlight (helix v/x, or a tagline sweep),
- // inclusive [lo, hi] — rendered-tag columns, so no conversion.
+ // inclusive [lo, hi], mapped from byte offsets to display cells.
//
// Every selection on screen paints in the LIVE theme (`th`) and not in
// `chrome`: a highlight is not attached to any geometry, it appears
@@ -13055,8 +13291,9 @@ pub const Pardes = struct {
// animation's frames.
if (pane.tag_edit and pane.tag_sel) {
const b = tagSelBounds(pane);
- var col: usize = b.lo;
- const end: usize = b.hi;
+ var col = file_pane.rawDisplayCol(tag, b.lo);
+ const hi = modal.nextGrapheme(tag, b.hi);
+ const end = file_pane.rawDisplayCol(tag, hi) -| 1;
while (col <= end and col < tw) : (col += 1) {
const cell = s.at(tx + @as(u16, @intCast(col)), tag_y);
cell.default = false;
@@ -13064,10 +13301,11 @@ pub const Pardes = struct {
cell.style.fg = .{ .rgb = th.sel_fg };
}
}
- // cursor while editing the tag: a rendered-tag column, verbatim
+ // cursor while editing the tag: byte offset mapped to its display cell
if (active and pane.tag_edit) {
// bar while typing, block for `:` normal mode (same rule as a body)
- if (pane.tag_col < tw) s.cursor = .{ .x = tx + pane.tag_col, .y = tag_y, .bar = pane.mode == .insert };
+ const col = file_pane.rawDisplayCol(tag, pane.tag_col);
+ if (col < tw) s.cursor = .{ .x = tx + @as(u16, @intCast(col)), .y = tag_y, .bar = pane.mode == .insert };
}
// A native PDF page uses the same backend-neutral pixel attachment as
@@ -13243,12 +13481,19 @@ pub const Pardes = struct {
while (vr + @as(i32, BOX_H) < @as(i32, r.h)) : (vr += 1) {
const w = pane.wrapAt(vr);
if (w.line < bnd.lo_row or w.line > bnd.hi_row) continue;
+ const visible_line = modal.lineSlice(body, @intCast(vr));
const cstart: i32 = if (w.line == bnd.lo_row)
- (if (pane.file != null) file_pane.displayOffset(pane, w.line, w.at, bnd.lo_col) else bnd.lo_col - w.at) + vpfx
+ (if (pane.file != null)
+ file_pane.displayOffset(pane, w.line, w.at, bnd.lo_col)
+ else
+ file_pane.lineDisplayOffset(visible_line, @intCast(@max(0, w.at)), @intCast(@max(0, bnd.lo_col)))) + vpfx
else
vpfx;
const cend: i32 = if (w.line == bnd.hi_row)
- (if (pane.file != null) file_pane.displayEndOffset(pane, w.line, w.at, bnd.hi_col) else bnd.hi_col - w.at) + vpfx
+ (if (pane.file != null)
+ file_pane.displayEndOffset(pane, w.line, w.at, bnd.hi_col)
+ else
+ file_pane.lineDisplayEndOffset(visible_line, @intCast(@max(0, w.at)), @intCast(@max(0, bnd.hi_col)))) + vpfx
else
@as(i32, tw) - 1;
var col: i32 = @max(cstart, vpfx);
@@ -13263,7 +13508,14 @@ pub const Pardes = struct {
if (hover_only) continue; // quiet preview preserves the source ink
const cw = pane.wrapRow(sr.row, sr.col);
const crow = cw.row + @as(i32, BOX_H);
- const ccol = (if (pane.file != null) file_pane.displayOffset(pane, sr.row, cw.at, sr.col) else sr.col - cw.at) + vpfx;
+ const ccol = (if (pane.file != null)
+ file_pane.displayOffset(pane, sr.row, cw.at, sr.col)
+ else
+ file_pane.lineDisplayOffset(
+ modal.lineSlice(body, @intCast(@max(0, cw.row))),
+ @intCast(@max(0, cw.at)),
+ @intCast(@max(0, sr.col)),
+ )) + vpfx;
if (crow >= BOX_H and crow < @as(i32, r.h) and ccol >= vpfx and ccol < tw) {
const cell = s.at(tx + @as(u16, @intCast(ccol)), body_y + @as(u16, @intCast(crow - BOX_H)));
cell.default = false;
@@ -13290,6 +13542,12 @@ pub const Pardes = struct {
// cells, so account for every expanded tab before the cursor.
const cx = if (pane.file != null)
@as(i32, config.PREFIX_W) + file_pane.displayOffset(pane, crow, cwp.at, ccol)
+ else if (pane.cur_pinned)
+ file_pane.lineDisplayOffset(
+ modal.lineSlice(body, @intCast(@max(0, cwp.row))),
+ @intCast(@max(0, cwp.at)),
+ @intCast(@max(0, ccol)),
+ )
else
ccol;
if (prow >= BOX_H and cx >= 0 and prow < r.h and cx < tw)
diff --git a/src/pdf_pane_integration_test.zig b/src/pdf_pane_integration_test.zig
index eaebae5e..b290616e 100644
--- a/src/pdf_pane_integration_test.zig
+++ b/src/pdf_pane_integration_test.zig
@@ -111,7 +111,8 @@ test "PdfSections Look follows the exact owning PDF, not an equal path" {
const p = try Pardes.init(gpa, .{ .file = path, .cols = 80, .rows = 28 });
defer p.deinit();
const first = p.panes[0].?;
- pdf_pane.openSections(p, 0);
+ // Bare normal-mode `f` on a PDF dispatches the PdfSections builtin.
+ p.update(.{ .key = .{ .cp = 'f', .text = "f" } });
const first_output_id = first.search_pane orelse return error.MissingPdfSectionsOutput;
const first_output = p.panes[first_output_id].?;
try std.testing.expect(pdf_pane.isSectionsOutput(first_output));
diff --git a/src/syntax.zig b/src/syntax.zig
index b7b9f32f..def11420 100644
--- a/src/syntax.zig
+++ b/src/syntax.zig
@@ -1,7 +1,7 @@
//! Tree-sitter syntax highlighting: one style byte per content byte, filled by
//! running each grammar's highlights.scm query (slurped at build time into the
//! ts_queries options module). Grammar set is tiered: `zig`; `minimal` (c, cpp,
-//! zig); and `full`, which adds ~23 languages lazily on first use.
+//! zig); and `full`, which adds ~24 languages lazily on first use.
const std = @import("std");
const config = @import("pardes_config");
const tracy = @import("tracy.zig");
@@ -173,6 +173,7 @@ fn synFor(name: []const u8) Syn {
.{ "string", .string },
.{ "character", .string },
.{ "number", .number },
+ .{ "numeric", .number },
.{ "float", .number },
.{ "boolean", .number },
.{ "keyword", .keyword },
@@ -239,3 +240,26 @@ test "tree-sitter allocator callbacks preserve and free exact allocations" {
try std.testing.expect(syntaxRealloc(live, 0) == null);
live = null;
}
+
+test "default full grammar set highlights Typst source" {
+ if (!enabled or !full_grammars) return;
+ start(std.testing.allocator);
+ defer stop();
+
+ const source = "// note\n#let answer = 42\n#let text = \"hello\"\n";
+ const styles = try highlightFileRange(std.testing.allocator, "paper.typst", source, 0, source.len);
+ defer std.testing.allocator.free(styles);
+
+ const comment_at = std.mem.indexOf(u8, source, "// note").?;
+ const keyword_at = std.mem.indexOf(u8, source, "let").?;
+ const number_at = std.mem.indexOf(u8, source, "42").?;
+ const string_at = std.mem.indexOf(u8, source, "\"hello\"").?;
+ try std.testing.expectEqual(Syn.comment, @as(Syn, @enumFromInt(styles[comment_at])));
+ try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(styles[keyword_at])));
+ try std.testing.expectEqual(Syn.number, @as(Syn, @enumFromInt(styles[number_at])));
+ try std.testing.expectEqual(Syn.string, @as(Syn, @enumFromInt(styles[string_at])));
+
+ const short_ext = try highlightFileRange(std.testing.allocator, "paper.typ", source, 0, source.len);
+ defer std.testing.allocator.free(short_ext);
+ try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(short_ext[keyword_at])));
+}