diff options
| -rw-r--r-- | build.zig | 1 | ||||
| -rw-r--r-- | build.zig.zon | 4 | ||||
| -rw-r--r-- | docs/helix-keys.md | 6 | ||||
| -rw-r--r-- | docs/lsp-evaluation.md | 2 | ||||
| -rw-r--r-- | src/config.zig | 7 | ||||
| -rw-r--r-- | src/file_pane.zig | 122 | ||||
| -rw-r--r-- | src/grammar_manifest.zig | 1 | ||||
| -rw-r--r-- | src/modal.zig | 418 | ||||
| -rw-r--r-- | src/normal_input.zig | 12 | ||||
| -rw-r--r-- | src/pardes.zig | 462 | ||||
| -rw-r--r-- | src/pdf_pane_integration_test.zig | 3 | ||||
| -rw-r--r-- | src/syntax.zig | 26 | ||||
| -rw-r--r-- | test/snapshots/badutf.golden | 2 | ||||
| -rw-r--r-- | test/snapshots/badutf.snap | 11 |
14 files changed, 788 insertions, 289 deletions
@@ -607,6 +607,7 @@ pub fn build(b: *std.Build) void { if (vaxis_mod) |vaxis| { vaxis.addImport("uucode", uucode_mod); root_mod.addImport("vaxis", vaxis); + hx_core_mod.addImport("vaxis", vaxis); } root_mod.addImport("uucode", uucode_mod); hx_core_mod.addImport("uucode", uucode_mod); diff --git a/build.zig.zon b/build.zig.zon index e6cdcc5c..9ccb686f 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -60,6 +60,10 @@ .url = "git+https://github.com/tree-sitter/tree-sitter-c#b780e47fc780ddc8da13afa35a3f4ed5c157823d", .hash = "tree_sitter_c-0.24.2-y5boS-ptQADHoCoVfjGT_nFtFQ5LbomIkW0fxG3_cmdB", }, + .ts_typst = .{ + .url = "git+https://github.com/uben0/tree-sitter-typst#13863ddcbaa7b68ee6221cea2e3143415e64aea4", + .hash = "N-V-__8AANAJVQBOabRzU-zDAdM20_2B2W2qk3-6Gvm1p1ZS", + }, .ts_zig = .{ .url = "git+https://github.com/maxxnino/tree-sitter-zig#a80a6e9be81b33b182ce6305ae4ea28e29211bd5", .hash = "N-V-__8AADLAWgBszUZPv5YTGyO5ZNyxNcuuR8MB3mIMJrZf", diff --git a/docs/helix-keys.md b/docs/helix-keys.md index 2aff75cd..da969574 100644 --- a/docs/helix-keys.md +++ b/docs/helix-keys.md @@ -37,8 +37,10 @@ layer: gap offsets + ranges over the flat text). Differential harness: cursor when the selection is implicit motion residue. Mode is reported "select" ONLY while `v` extend mode is on (helix keeps mode normal for x/%/mi-created selections). -- **Columns are byte offsets** (`cur_col`). ASCII-only harness cases - sidestep this; real UTF-8 parity is out of scope. +- **Stored columns are UTF-8 byte offsets** (`cur_col`), but every modal + cursor/range endpoint is snapped to an extended-grapheme boundary. Rendering, + mouse input and vertical motion translate those offsets through terminal-cell + widths, so combining sequences and wide glyphs remain single cursor cells. - **Terminal panes: shell output is immutable.** Edit ops only ever drop/alter typed insertion runs. All "To implement" edit ops follow the same rule (motion/selection parts work on the full motion surface; the diff --git a/docs/lsp-evaluation.md b/docs/lsp-evaluation.md index 60e1442f..55049ec6 100644 --- a/docs/lsp-evaluation.md +++ b/docs/lsp-evaluation.md @@ -118,7 +118,7 @@ second process, and its weakest number is a cache that does not exist rather than a limit that cannot be lifted. The one thing that should override that: **pardes ships tree-sitter grammars for -26 languages.** C is Zig-only forever; B is the only implementation that will +27 languages.** C is Zig-only forever; B is the only implementation that will ever answer `gd` in a Rust or Go buffer. If language intelligence is meant to follow the syntax highlighting, B is the strategic choice and its 721 lines are the cheapest multi-language client anyone will write. diff --git a/src/config.zig b/src/config.zig index 9f18f75e..f58716e5 100644 --- a/src/config.zig +++ b/src/config.zig @@ -581,7 +581,10 @@ pub const MINH: u16 = 3; /// definition of "the word under the cursor" for Look, Execute, the tag chord /// and every search result row. pub fn isFileChar(c: u8) bool { - return std.ascii.isAlphanumeric(c) or switch (c) { + // Non-ASCII bytes belong to their UTF-8 word as a unit. Bounds are byte + // offsets, so accepting every high byte keeps Unicode paths/identifiers + // intact instead of returning a slice through one codepoint. + return c >= 0x80 or std.ascii.isAlphanumeric(c) or switch (c) { '.', '-', '+', '/', ':', '@', '_', '~' => true, else => false, }; @@ -677,7 +680,7 @@ pub const image_exts = [_][]const u8{ ".png", ".jpg", ".jpeg", ".gif", ".bmp", " /// token for them would be a divergence nothing asked for. pub const comment_token_default = "#"; pub const comment_tokens: []const struct { exts: []const []const u8, token: []const u8 } = &.{ - .{ .token = "//", .exts = &.{ ".zig", ".zon", ".c", ".h", ".cpp", ".cc", ".cxx", ".hpp", ".hh", ".hxx", ".rs", ".go", ".java", ".scala", ".sc", ".kt", ".kts", ".cs", ".csx", ".php", ".pas", ".pp", ".p", ".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".swift", ".dart" } }, + .{ .token = "//", .exts = &.{ ".zig", ".zon", ".c", ".h", ".cpp", ".cc", ".cxx", ".hpp", ".hh", ".hxx", ".rs", ".go", ".java", ".scala", ".sc", ".kt", ".kts", ".cs", ".csx", ".php", ".pas", ".pp", ".p", ".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".typ", ".typst", ".swift", ".dart" } }, .{ .token = "#", .exts = &.{ ".py", ".pyw", ".sh", ".bash", ".zsh", ".rb", ".rake", ".ex", ".exs", ".ps1", ".psm1", ".psd1", ".pl", ".pm", ".r", ".jl", ".nix", ".toml", ".yaml", ".yml", ".cmake", ".mk", ".tf" } }, .{ .token = "--", .exts = &.{ ".lua", ".hs", ".lhs", ".elm", ".sql", ".adb", ".ads", ".ada" } }, .{ .token = ";", .exts = &.{ ".clj", ".cljs", ".cljc", ".edn", ".el", ".lisp", ".scm", ".asm", ".s" } }, diff --git a/src/file_pane.zig b/src/file_pane.zig index cf2b9571..abf37448 100644 --- a/src/file_pane.zig +++ b/src/file_pane.zig @@ -4,6 +4,7 @@ //! number gutter and the syntax recolor). The rest of a file pane's behaviour //! is the pane machinery in pardes.zig, which does not care what kind it is. const std = @import("std"); +const vaxis = @import("vaxis"); const pardes = @import("pardes.zig"); const config = @import("config.zig"); const Pardes = pardes.Pardes; @@ -82,13 +83,23 @@ pub fn dumpPane( }; } +pub fn graphemeDisplayWidth(grapheme: []const u8) usize { + if (std.mem.eql(u8, grapheme, "\t")) return config.tab_width; + return @max(1, @as(usize, vaxis.gwidth.gwidth(grapheme, .unicode))); +} + pub fn byteDisplayWidth(byte: u8) usize { return if (byte == '\t') config.tab_width else 1; } pub fn displayWidth(text: []const u8) usize { var width: usize = 0; - for (text) |byte| width +|= byteDisplayWidth(byte); + var at: usize = 0; + while (at < text.len) { + const end = modal.nextGrapheme(text, at); + width +|= graphemeDisplayWidth(text[at..end]); + at = end; + } return width; } @@ -96,10 +107,13 @@ pub fn displayWidth(text: []const u8) usize { /// maps back to that one tab byte. pub fn byteAtDisplay(text: []const u8, display_col: usize) usize { var col: usize = 0; - for (text, 0..) |byte, i| { - const next = col +| byteDisplayWidth(byte); - if (display_col < next) return i; + var at: usize = 0; + while (at < text.len) { + const end = modal.nextGrapheme(text, at); + const next = col +| graphemeDisplayWidth(text[at..end]); + if (display_col < next) return at; col = next; + at = end; } return text.len; } @@ -107,8 +121,8 @@ pub fn byteAtDisplay(text: []const u8, display_col: usize) usize { /// File cursor columns may live past EOL. Tabs expand before that boundary; /// every virtual column after it remains one screen cell. pub fn rawDisplayCol(line_text: []const u8, raw_col: usize) usize { - const bounded = @min(raw_col, line_text.len); - return displayWidth(line_text[0..bounded]) +| (raw_col - bounded); + const bounded = modal.graphemeStart(line_text, @min(raw_col, line_text.len)); + return displayWidth(line_text[0..bounded]) +| (raw_col -| line_text.len); } pub fn rawAtDisplay(line_text: []const u8, display_col: usize) usize { @@ -117,6 +131,27 @@ pub fn rawAtDisplay(line_text: []const u8, display_col: usize) usize { return byteAtDisplay(line_text, display_col); } +pub fn byteAtDisplayFrom(line_text: []const u8, from_raw: usize, display_col: usize) usize { + if (from_raw >= line_text.len) return from_raw +| display_col; + const from = modal.graphemeStart(line_text, from_raw); + return from +| rawAtDisplay(line_text[from..], display_col); +} + +pub fn lineDisplayOffset(line_text: []const u8, from_raw: usize, to_raw: usize) i32 { + const from_display = rawDisplayCol(line_text, from_raw); + const to_display = rawDisplayCol(line_text, to_raw); + if (to_display >= from_display) return @intCast(to_display - from_display); + return -@as(i32, @intCast(from_display - to_display)); +} + +pub fn lineDisplayEndOffset(line_text: []const u8, from_raw: usize, at_raw: usize) i32 { + const start = lineDisplayOffset(line_text, from_raw, at_raw); + if (at_raw >= line_text.len) return start; + const at = modal.graphemeStart(line_text, at_raw); + const end = modal.nextGrapheme(line_text, at); + return start + @as(i32, @intCast(graphemeDisplayWidth(line_text[at..end]))) - 1; +} + pub fn sourceLine(pane: *const Pane, row: i32) []const u8 { const f = pane.file orelse return ""; if (row < 0) return ""; @@ -127,51 +162,60 @@ pub fn displayOffset(pane: *const Pane, row: i32, from_raw: i32, to_raw: i32) i3 const line_text = sourceLine(pane, row); const from: usize = @intCast(@max(0, from_raw)); const to: usize = @intCast(@max(0, to_raw)); - const from_display = rawDisplayCol(line_text, from); - const to_display = rawDisplayCol(line_text, to); - if (to_display >= from_display) return @intCast(to_display - from_display); - return -@as(i32, @intCast(from_display - to_display)); + return lineDisplayOffset(line_text, from, to); } pub fn displayEndOffset(pane: *const Pane, row: i32, from_raw: i32, at_raw: i32) i32 { - const start = displayOffset(pane, row, from_raw, at_raw); const line_text = sourceLine(pane, row); - const at: usize = @intCast(@max(0, at_raw)); - if (at >= line_text.len) return start; - return start + @as(i32, @intCast(byteDisplayWidth(line_text[at]))) - 1; + return lineDisplayEndOffset(line_text, @intCast(@max(0, from_raw)), @intCast(@max(0, at_raw))); } pub fn byteAtRowDisplay(pane: *const Pane, row: i32, from_raw: i32, display_col: i32) i32 { const line_text = sourceLine(pane, row); - const from: usize = @intCast(@max(0, from_raw)); - const display: usize = @intCast(@max(0, display_col)); - if (from >= line_text.len) return @intCast(from +| display); - return @intCast(from +| rawAtDisplay(line_text[from..], display)); + return @intCast(byteAtDisplayFrom(line_text, @intCast(@max(0, from_raw)), @intCast(@max(0, display_col)))); } -/// Convert between rendered pane rows (whose file body includes PREFIX_W) -/// and source-byte columns. Non-file rows are already in byte coordinates. +/// Convert between rendered cells and UTF-8 byte columns. Tag rows always +/// need grapheme conversion; file body rows additionally skip PREFIX_W. pub fn renderedLineByteCol(pane: *const Pane, row: i32, line_text: []const u8, display_col: usize) usize { - if (pane.file == null or row < pardes.BOX_H) return display_col; + if (row < pardes.BOX_H) return rawAtDisplay(line_text, display_col); + if (pane.file == null) return rawAtDisplay(line_text, display_col); const prefix = @min(@as(usize, config.PREFIX_W), line_text.len); if (display_col <= prefix) return display_col; return prefix +| rawAtDisplay(line_text[prefix..], display_col - prefix); } pub fn renderedLineDisplayCol(pane: *const Pane, row: i32, line_text: []const u8, byte_col: usize) usize { - if (pane.file == null or row < pardes.BOX_H) return byte_col; + if (row < pardes.BOX_H) return rawDisplayCol(line_text, byte_col); + if (pane.file == null) return rawDisplayCol(line_text, byte_col); const prefix = @min(@as(usize, config.PREFIX_W), line_text.len); if (byte_col <= prefix) return byte_col; return prefix +| rawDisplayCol(line_text[prefix..], byte_col - prefix); } +test "display columns map complete Unicode graphemes" { + const text = "é界e\u{301}x"; + try std.testing.expectEqual(@as(usize, 5), displayWidth(text)); + try std.testing.expectEqual(@as(usize, 0), byteAtDisplay(text, 0)); + try std.testing.expectEqual(@as(usize, 2), byteAtDisplay(text, 1)); + try std.testing.expectEqual(@as(usize, 2), byteAtDisplay(text, 2)); + try std.testing.expectEqual(@as(usize, 5), byteAtDisplay(text, 3)); + try std.testing.expectEqual(@as(usize, 8), byteAtDisplay(text, 4)); + try std.testing.expectEqual(text.len, byteAtDisplay(text, 5)); + try std.testing.expectEqual(@as(usize, 3), rawDisplayCol(text, 5)); + try std.testing.expectEqual(@as(usize, 5), rawAtDisplay(text, 3)); + try std.testing.expectEqual(@as(usize, 2), graphemeDisplayWidth("👩\u{200d}🚀")); +} + fn fitEnd(text: []const u8, start: usize, width: usize) usize { var end = start; var used: usize = 0; - while (end < text.len) : (end += 1) { - const next = used +| byteDisplayWidth(text[end]); - if (next > width) return if (end == start) start + 1 else end; - used = next; + while (end < text.len) { + const next_end = modal.nextGrapheme(text, end); + const next_used = used +| graphemeDisplayWidth(text[end..next_end]); + if (next_used > width) return if (end == start) next_end else end; + used = next_used; + end = next_end; } return end; } @@ -253,7 +297,7 @@ pub fn textOffset(pane: *Pane, text: []const u8, cursor: modal.Cursor) usize { const row = @min(cursor.row, index.len - 1); const start = index[row]; const end = if (row + 1 < index.len) index[row + 1] - 1 else text.len; - return @min(start + cursor.col, end); + return start + modal.graphemeStart(text[start..end], @min(cursor.col, end - start)); } pub fn textLineStart(pane: *Pane, text: []const u8, row: usize) usize { @@ -274,7 +318,9 @@ pub fn textPosition(pane: *Pane, text: []const u8, offset: usize) modal.Cursor { return std.math.order(key, item); } }.cmp) - 1; - return .{ .row = row, .col = bounded - index[row] }; + const start = index[row]; + const end = if (row + 1 < index.len) index[row + 1] - 1 else text.len; + return .{ .row = row, .col = modal.graphemeStart(text[start..end], @min(bounded - start, end - start)) }; } pub fn open(p: *Pardes, id: usize, path: []const u8, line: usize) !*Pane { @@ -577,17 +623,14 @@ fn fillBody(dst: ?[]u8, pane: *Pane, f: *State, width: usize, record_wrap: bool) if (dst) |out| @memcpy(out[written..][0..prefix.len], prefix); written += prefix.len; - // A byte cut, like the hscroll one below, and it can land inside a - // multi-byte glyph for the same reason: columns here are BYTES. - // Surface.print decodes by hand and emits U+FFFD per undecodable - // byte, so a split glyph renders as a replacement char rather than - // panicking — see test/snapshots/badutf.snap. It cannot overflow - // the pane either: a UTF-8 sequence is never fewer bytes than the - // cells it draws in. + // Wrap and horizontal-scroll cuts are always grapheme boundaries. + // Source columns remain byte offsets, while widths are terminal + // cells; keeping the conversion here prevents a view operation + // from manufacturing malformed UTF-8. const end = if (width == 0) text.len else fitEnd(text, at, width); const take = end - at; const cut = if (pane.hscroll > 0 and width == 0) - @min(@as(usize, @intCast(pane.hscroll)), take) + modal.graphemeStart(text[at..end], @min(@as(usize, @intCast(pane.hscroll)), take)) else 0; const shown = text[at + cut .. end]; @@ -690,10 +733,12 @@ pub fn recolorSyntax(p: *Pardes, pane: *Pane, f: *State, r: pardes.Rect, tx: u16 else line.len; } + hs = modal.graphemeStart(line, @min(hs, line.len)); var c: usize = 0; var screen_c: usize = 0; - while (hs + c < limit and config.PREFIX_W + screen_c < tw) : (c += 1) { - const cells = byteDisplayWidth(line[hs + c]); + while (hs + c < limit and config.PREFIX_W + screen_c < tw) { + const grapheme_end = @min(limit, modal.nextGrapheme(line, hs + c)); + const cells = graphemeDisplayWidth(line[hs + c .. grapheme_end]); const idx = base + hs + c; if (idx >= f.highlight_start) { const hidx = idx - f.highlight_start; @@ -710,6 +755,7 @@ pub fn recolorSyntax(p: *Pardes, pane: *Pane, f: *State, r: pardes.Rect, tx: u16 } } screen_c += cells; + c = grapheme_end - hs; } } } diff --git a/src/grammar_manifest.zig b/src/grammar_manifest.zig index bd89e0a0..101cd6f6 100644 --- a/src/grammar_manifest.zig +++ b/src/grammar_manifest.zig @@ -40,5 +40,6 @@ pub const all = [_]Grammar{ .{ .name = "ruby", .dep = "ts_ruby", .exts = &.{ ".rb", ".rake" }, .tier = .full, .scanner = true }, .{ .name = "rust", .dep = "ts_rust", .exts = &.{".rs"}, .tier = .full, .scanner = true }, .{ .name = "scala", .dep = "ts_scala", .exts = &.{ ".scala", ".sc" }, .tier = .full, .scanner = true }, + .{ .name = "typst", .dep = "ts_typst", .exts = &.{ ".typ", ".typst" }, .tier = .full, .scanner = true, .query = "queries/typst/highlights.scm" }, .{ .name = "zig", .dep = "ts_zig", .exts = &.{ ".zig", ".zon" }, .tier = .zig }, }; diff --git a/src/modal.zig b/src/modal.zig index 6ab7ee35..a98cfeb4 100644 --- a/src/modal.zig +++ b/src/modal.zig @@ -1,12 +1,14 @@ const std = @import("std"); +const uucode = @import("uucode"); // Modal-editing text math, kept free of vaxis/ghostty so it can be unit-tested // in isolation (see the `unit-test` build step). main.zig wires this onto the // pane's cursor + (for file panes) its content. // -// The cursor sits ON a character: col is a char index in [0, line.len]; col == -// line.len means "on the line terminator / after the last char". Motions are -// written to land on real characters; main.zig clamps for display. +// The cursor sits ON a grapheme: col is its UTF-8 byte offset in +// [0, line.len]; col == line.len means "on the line terminator / after the +// last grapheme". Motions never leave a cursor in the middle of UTF-8 or an +// extended grapheme cluster. pub const Cursor = struct { row: usize = 0, @@ -21,23 +23,57 @@ pub const Cursor = struct { // non-ws, ws = space/tab/newline). pub const Kind = enum { word, punct, ws }; +fn codepointAt(text: []const u8, off: usize) u21 { + if (off >= text.len) return 0xFFFD; + const n = std.unicode.utf8ByteSequenceLength(text[off]) catch return 0xFFFD; + if (off + n > text.len) return 0xFFFD; + return std.unicode.utf8Decode(text[off .. off + n]) catch 0xFFFD; +} + +fn isUnicodeWhitespace(cp: u21) bool { + if (cp == ' ' or (cp >= '\t' and cp <= '\r') or cp == 0x85) return true; + return switch (uucode.get(.general_category, cp)) { + .separator_space, .separator_line, .separator_paragraph => true, + else => false, + }; +} + +fn kindOfCodepoint(cp: u21) Kind { + if (isUnicodeWhitespace(cp)) return .ws; + if (cp == '_') return .word; + return switch (uucode.get(.general_category, cp)) { + .letter_uppercase, + .letter_lowercase, + .letter_titlecase, + .letter_modifier, + .letter_other, + .mark_nonspacing, + .mark_spacing_combining, + .mark_enclosing, + .number_decimal_digit, + .number_letter, + .number_other, + .punctuation_connector, + => .word, + else => .punct, + }; +} + pub fn kindOf(c: u8) Kind { - if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws; - if (std.ascii.isAlphanumeric(c) or c == '_') return .word; - return .punct; + return kindOfCodepoint(c); } // "long word" (W/B/E): only whitespace separates; punct is part of a word. -fn kindOfLong(c: u8) Kind { - if (c == ' ' or c == '\t' or c == '\n' or c == '\r') return .ws; - return .word; +fn kindOfLong(cp: u21) Kind { + return if (isUnicodeWhitespace(cp)) .ws else .word; } fn kindAt(lines: []const []const u8, c: Cursor, long: bool) Kind { if (c.row >= lines.len) return .ws; const line = lines[c.row]; if (c.col >= line.len) return .ws; // line terminator / EOF = whitespace - return if (long) kindOfLong(line[c.col]) else kindOf(line[c.col]); + const cp = codepointAt(line, graphemeStart(line, c.col)); + return if (long) kindOfLong(cp) else kindOfCodepoint(cp); } fn lineLenOf(lines: []const []const u8, row: usize) usize { @@ -51,7 +87,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool { if (c.row >= lines.len) return false; const llen = lineLenOf(lines, c.row); if (c.col < llen) { - c.col += 1; + c.col = nextGrapheme(lines[c.row], c.col); return true; } // at the newline: move to next line start @@ -65,7 +101,7 @@ fn stepFwd(lines: []const []const u8, c: *Cursor) bool { fn stepBwd(lines: []const []const u8, c: *Cursor) bool { if (c.col > 0) { - c.col -= 1; + c.col = prevGrapheme(lines[c.row], c.col); return true; } if (c.row == 0) return false; @@ -83,7 +119,7 @@ fn atEof(lines: []const []const u8, c: Cursor) bool { pub fn firstNonWs(line: []const u8) usize { var i: usize = 0; - while (i < line.len and (line[i] == ' ' or line[i] == '\t')) i += 1; + while (i < line.len and isUnicodeWhitespace(codepointAt(line, i))) i = nextGrapheme(line, i); return i; } @@ -95,7 +131,7 @@ pub fn lineStart(c: Cursor) Cursor { pub fn lineEnd(lines: []const []const u8, c: Cursor) Cursor { const llen = lineLenOf(lines, c.row); - return .{ .row = c.row, .col = if (llen == 0) 0 else llen - 1 }; + return .{ .row = c.row, .col = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen) }; } pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor { @@ -107,28 +143,33 @@ pub fn firstNonWsOf(lines: []const []const u8, c: Cursor) Cursor { // ---- char/line motions ---- -pub fn charLeft(c: Cursor) Cursor { - return .{ .row = c.row, .col = if (c.col > 0) c.col - 1 else 0 }; +pub fn charLeft(lines: []const []const u8, c: Cursor) Cursor { + if (c.row >= lines.len) return .{ .row = c.row, .col = 0 }; + return .{ .row = c.row, .col = prevGrapheme(lines[c.row], @min(c.col, lines[c.row].len)) }; } pub fn charRight(lines: []const []const u8, c: Cursor) Cursor { const llen = lineLenOf(lines, c.row); - const last = if (llen == 0) 0 else llen - 1; - return .{ .row = c.row, .col = if (c.col < last) c.col + 1 else last }; + const last = if (llen == 0) 0 else prevGrapheme(lines[c.row], llen); + return .{ .row = c.row, .col = if (c.col < last) @min(nextGrapheme(lines[c.row], c.col), last) else last }; +} + +fn clampLineCol(line: []const u8, col: usize) usize { + if (line.len == 0) return 0; + const last = prevGrapheme(line, line.len); + return graphemeStart(line, @min(col, last)); } pub fn lineDown(lines: []const []const u8, c: Cursor) Cursor { const nr = if (c.row + 1 < lines.len) c.row + 1 else c.row; const llen = lineLenOf(lines, nr); - const last = if (llen == 0) 0 else llen - 1; - return .{ .row = nr, .col = if (c.col < last) c.col else last }; + return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) }; } pub fn lineUp(lines: []const []const u8, c: Cursor) Cursor { const nr = if (c.row > 0) c.row - 1 else c.row; const llen = lineLenOf(lines, nr); - const last = if (llen == 0) 0 else llen - 1; - return .{ .row = nr, .col = if (c.col < last) c.col else last }; + return .{ .row = nr, .col = if (llen == 0) 0 else clampLineCol(lines[nr], c.col) }; } // ---- word motions ---- @@ -205,19 +246,22 @@ pub fn gotoLast(lines: []const []const u8) Cursor { } // the character the cursor sits on; line terminators / EOF read as '\n'. -fn charAt(lines: []const []const u8, c: Cursor) u8 { +fn codepointAtCursor(lines: []const []const u8, c: Cursor) u21 { if (c.row >= lines.len) return '\n'; const line = lines[c.row]; if (c.col >= line.len) return '\n'; - return line[c.col]; + return codepointAt(line, graphemeStart(line, c.col)); +} + +fn charAt(lines: []const []const u8, c: Cursor) u8 { + const cp = codepointAtCursor(lines, c); + return if (cp <= 0x7f) @intCast(cp) else 0; } // `f`/`F`/`t`/`T`: the nth occurrence of `ch` after/before the cursor, across // line boundaries (helix: not confined to the line). `till` stops one position // short of the hit. Returns null (no move) when there aren't n occurrences. pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: bool, n: usize) ?Cursor { - if (ch > 0x7f) return null; // ponytail: ASCII targets only (byte columns) - const target: u8 = @intCast(ch); var p = clampToChar(lines, c); var left = if (n == 0) 1 else n; while (left > 0) { @@ -226,7 +270,7 @@ pub fn findChar(lines: []const []const u8, c: Cursor, ch: u21, fwd: bool, till: } else { if (!stepBwd(lines, &p)) return null; } - if (charAt(lines, p) == target) left -= 1; + if (codepointAtCursor(lines, p) == ch) left -= 1; } if (till) { if (fwd) _ = stepBwd(lines, &p) else _ = stepFwd(lines, &p); @@ -369,16 +413,32 @@ pub fn wordRange(lines: []const []const u8, c0: Cursor, long: bool, around: bool const k = kindAt(lines, c, long); if (k == .ws) return null; var lo = c.col; - while (lo > 0 and kindAt(lines, .{ .row = c.row, .col = lo - 1 }, long) == k) lo -= 1; + while (lo > 0) { + const prev = prevGrapheme(line, lo); + if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != k) break; + lo = prev; + } var hi = c.col; - while (hi + 1 < line.len and kindAt(lines, .{ .row = c.row, .col = hi + 1 }, long) == k) hi += 1; + while (true) { + const next = nextGrapheme(line, hi); + if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != k) break; + hi = next; + } if (around) { var h2 = hi; - while (h2 + 1 < line.len and (line[h2 + 1] == ' ' or line[h2 + 1] == '\t')) h2 += 1; + while (true) { + const next = nextGrapheme(line, h2); + if (next >= line.len or kindAt(lines, .{ .row = c.row, .col = next }, long) != .ws) break; + h2 = next; + } if (h2 != hi) { hi = h2; } else { - while (lo > 0 and (line[lo - 1] == ' ' or line[lo - 1] == '\t')) lo -= 1; + while (lo > 0) { + const prev = prevGrapheme(line, lo); + if (kindAt(lines, .{ .row = c.row, .col = prev }, long) != .ws) break; + lo = prev; + } } } return .{ .a = .{ .row = c.row, .col = lo }, .b = .{ .row = c.row, .col = hi } }; @@ -403,7 +463,7 @@ pub fn paragraphRange(lines: []const []const u8, c0: Cursor, around: bool) ?Rang } } const llen = lineLenOf(lines, r1); - return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else llen - 1 } }; + return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else prevGrapheme(lines[r1], llen) } }; } // ---- helpers used by motions + main.zig ---- @@ -415,7 +475,7 @@ pub fn clampToChar(lines: []const []const u8, c: Cursor) Cursor { } const llen = lineLenOf(lines, c.row); if (llen == 0) return .{ .row = c.row, .col = 0 }; - return .{ .row = c.row, .col = @min(c.col, llen - 1) }; + return .{ .row = c.row, .col = clampLineCol(lines[c.row], c.col) }; } pub fn lineCount(content: []const u8) usize { @@ -465,8 +525,9 @@ pub fn insertAt(alloc: std.mem.Allocator, content: []const u8, c: Cursor, text: pub fn deleteChar(alloc: std.mem.Allocator, content: []const u8, c: Cursor) ![]u8 { const line = lineSlice(content, c.row); if (c.col >= line.len) return alloc.dupe(u8, content); - const off = lineStartOffset(content, c.row) + c.col; - return spliceAlloc(alloc, content, off, off + 1, ""); + const col = graphemeStart(line, c.col); + const line_start = lineStartOffset(content, c.row); + return spliceAlloc(alloc, content, line_start + col, line_start + nextGrapheme(line, col), ""); } // delete whole lines [r0, r1] inclusive (the line content + their terminators). @@ -520,9 +581,13 @@ fn rangeBytes(content: []const u8, a: Cursor, b: Cursor) struct { s: usize, e: u lo = b; hi = a; } - const s = lineStartOffset(content, lo.row) + @min(lo.col, lineSlice(content, lo.row).len); - var e = lineStartOffset(content, hi.row) + @min(hi.col, lineSlice(content, hi.row).len); - if (e < content.len) e += 1; // include the char under the head + const lo_line = lineSlice(content, lo.row); + const hi_line = lineSlice(content, hi.row); + const lo_col = graphemeStart(lo_line, @min(lo.col, lo_line.len)); + const hi_col = graphemeStart(hi_line, @min(hi.col, hi_line.len)); + const s = lineStartOffset(content, lo.row) + lo_col; + var e = lineStartOffset(content, hi.row) + hi_col; + if (e < content.len) e = nextGrapheme(content, e); // include the grapheme under the head return .{ .s = s, .e = @max(s, e) }; } @@ -593,10 +658,22 @@ pub fn replaceRange(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: // `r<ch>`: overwrite every char in the inclusive range [a, b] with `ch` — // NEWLINES TOO (helix replace maps every grapheme, so `xrz` joins lines). -pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u8) ![]u8 { +pub fn replaceChars(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: Cursor, ch: u21) ![]u8 { const r = rangeBytes(content, a, b); - const out = try alloc.dupe(u8, content); - for (out[r.s..r.e]) |*p| p.* = ch; + var encoded: [4]u8 = undefined; + const encoded_len = try std.unicode.utf8Encode(ch, &encoded); + var count: usize = 0; + var at = r.s; + while (at < r.e) : (count += 1) at = nextGrapheme(content, at); + const replacement_len = count * encoded_len; + const out = try alloc.alloc(u8, content.len - (r.e - r.s) + replacement_len); + @memcpy(out[0..r.s], content[0..r.s]); + var write = r.s; + for (0..count) |_| { + @memcpy(out[write..][0..encoded_len], encoded[0..encoded_len]); + write += encoded_len; + } + @memcpy(out[write..], content[r.e..]); return out; } @@ -729,9 +806,9 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C // // Gap-offset ranges over the FLAT buffer, ported faithfully from // helix-core/src/movement.rs + selection.rs @ 278b24389 (the genizah -// checkout). Positions are gap offsets 0..=text.len — "char indices" in -// helix terms, bytes here (ASCII-exact; UTF-8 stepped by sequence, matching -// the byte-column convention of the rest of pardes). A range with +// checkout). Positions are UTF-8 gap offsets 0..=text.len. Stored columns +// remain byte offsets, but every range boundary is an extended-grapheme +// boundary. A range with // head > anchor selects [anchor, head) with the block cursor ON head-1; // head < anchor selects [head, anchor) with the cursor ON head. The // differential suite (test/hxcases, `zig build hxdiff`) pins every behavior @@ -739,19 +816,58 @@ pub fn deleteSpan(alloc: std.mem.Allocator, content: []const u8, a: Cursor, b: C pub const HxRange = struct { anchor: usize, head: usize }; -/// one grapheme forward (UTF-8 sequence step), clamped at text.len +/// The grapheme containing `off`, or text.len at EOF. This is also the repair +/// path for stale/external byte columns that happen to point into UTF-8. +pub fn graphemeStart(text: []const u8, off: usize) usize { + const bounded = @min(off, text.len); + if (bounded == text.len) return text.len; + var it = uucode.grapheme.utf8Iterator(text); + while (it.nextGrapheme()) |g| { + if (bounded < g.end) return g.start; + } + return text.len; +} + +/// one extended grapheme forward, clamped at text.len pub fn nextGrapheme(text: []const u8, off: usize) usize { if (off >= text.len) return text.len; - var o = off + 1; - while (o < text.len and (text[o] & 0xC0) == 0x80) o += 1; - return o; + // The editor's own offsets are already boundaries. Keep the overwhelmingly + // common ASCII path O(1); only repair a continuation-byte input here. + if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1; + var start = off; + while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1; + if (start != off) start = graphemeStart(text, off); + var it = uucode.grapheme.utf8Iterator(text[start..]); + const g = it.nextGrapheme() orelse return @min(start + 1, text.len); + return start + g.end; } pub fn prevGrapheme(text: []const u8, off: usize) usize { - if (off == 0) return 0; - var o = off - 1; - while (o > 0 and (text[o] & 0xC0) == 0x80) o -= 1; - return o; + var bounded = @min(off, text.len); + if (bounded < text.len) { + const repaired = graphemeStart(text, bounded); + if (repaired < bounded) return repaired; + bounded = repaired; + } + if (bounded == 0) return 0; + if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1; + // Graphemes cannot cross a line break. Restrict the forward segmentation + // needed for a reverse step to the current line instead of rescanning the + // complete buffer. + const line_start = if (std.mem.lastIndexOfScalar(u8, text[0 .. bounded - 1], '\n')) |nl| nl + 1 else 0; + if (line_start == bounded) return bounded - 1; // the newline is its own editor cell + var it = uucode.grapheme.utf8Iterator(text[line_start..bounded]); + var last = line_start; + while (it.nextGrapheme()) |g| last = line_start + g.start; + return last; +} + +/// Byte offset of the zero-based grapheme column, clamped to the line end. +pub fn graphemeAtColumn(text: []const u8, column: usize) usize { + var at: usize = 0; + var col: usize = 0; + while (at < text.len and col < column) : (col += 1) at = nextGrapheme(text, at); + return at; } /// the block cursor cell of a range (helix Range::cursor) @@ -792,14 +908,16 @@ pub fn hxLineEndIdx(text: []const u8, line: usize) usize { /// gap offset -> (row, col) cell pub fn hxPos(text: []const u8, off: usize) Cursor { - const o = @min(off, text.len); + const bounded = @min(off, text.len); // The line start is the byte after the last '\n' BEFORE off, which is the // same number lineStartOffset(text, row) walks the whole prefix to reach — // one backward scan of a single line instead of a second pass over // everything above the cursor. On a multi-MB buffer that second pass was // most of what a keystroke cost. - const s = if (std.mem.lastIndexOfScalar(u8, text[0..o], '\n')) |nl| nl + 1 else 0; - return .{ .row = hxLineOf(text, off), .col = o - s }; + const s = if (std.mem.lastIndexOfScalar(u8, text[0..bounded], '\n')) |nl| nl + 1 else 0; + const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; + const col = graphemeStart(text[s..e], @min(bounded - s, e - s)); + return .{ .row = hxLineOf(text, s + col), .col = col }; } /// (row, col) -> clamped gap offset; col == line length lands ON the '\n' @@ -809,7 +927,8 @@ pub fn hxOff(text: []const u8, c: Cursor) usize { // hxLineEndIdx(text, row) inlined: it starts by walking to `row` again, // and we are already standing there const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; - return @min(s + c.col, e); + const raw = @min(c.col, e - s); + return s + graphemeStart(text[s..e], raw); } pub const WordTarget = enum { @@ -827,35 +946,36 @@ pub const WordTarget = enum { // that distinction is load-bearing in reached_target. const HxCat = enum { word, punct, ws, eol }; -fn hxCat(b: u8) HxCat { - if (b == '\n' or b == '\r') return .eol; - if (b == ' ' or b == '\t' or b == 0x0b or b == 0x0c) return .ws; - if (std.ascii.isAlphanumeric(b) or b == '_' or b >= 0x80) return .word; - return .punct; +fn hxCatAt(text: []const u8, off: usize) HxCat { + if (off >= text.len) return .eol; + const cp = codepointAt(text, off); + if (cp == '\n' or cp == '\r') return .eol; + return switch (kindOfCodepoint(cp)) { + .word => .word, + .punct => .punct, + .ws => .ws, + }; } -fn hxIsWs(b: u8) bool { // Rust char::is_whitespace (includes line endings) - const c = hxCat(b); +fn hxIsWs(c: HxCat) bool { // Rust char::is_whitespace (includes line endings) return c == .ws or c == .eol; } -fn hxIsWordBoundary(a: u8, b: u8) bool { - return hxCat(a) != hxCat(b); +fn hxIsWordBoundary(a: HxCat, b: HxCat) bool { + return a != b; } -fn hxIsLongBoundary(a: u8, b: u8) bool { - const ca = hxCat(a); - const cb = hxCat(b); - if ((ca == .word and cb == .punct) or (ca == .punct and cb == .word)) return false; - return ca != cb; +fn hxIsLongBoundary(a: HxCat, b: HxCat) bool { + if ((a == .word and b == .punct) or (a == .punct and b == .word)) return false; + return a != b; } -fn hxReached(target: WordTarget, prev: u8, next: u8) bool { +fn hxReached(target: WordTarget, prev: HxCat, next: HxCat) bool { return switch (target) { - .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)), - .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol), - .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (hxCat(next) == .eol or !hxIsWs(next)), - .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or hxCat(next) == .eol), + .next_word_start, .prev_word_end => hxIsWordBoundary(prev, next) and (next == .eol or !hxIsWs(next)), + .next_word_end, .prev_word_start => hxIsWordBoundary(prev, next) and (!hxIsWs(prev) or next == .eol), + .next_long_word_start, .prev_long_word_end => hxIsLongBoundary(prev, next) and (next == .eol or !hxIsWs(next)), + .next_long_word_end, .prev_long_word_start => hxIsLongBoundary(prev, next) and (!hxIsWs(prev) or next == .eol), }; } @@ -896,45 +1016,35 @@ fn hxRangeToTarget(text: []const u8, target: WordTarget, origin: HxRange, is_pre var anchor = origin.anchor; var head = origin.head; var it = origin.head; - var prev_ch: ?u8 = if (is_prev) - (if (it < text.len) text[it] else null) + var prev_cat: ?HxCat = if (is_prev) + (if (it < text.len) hxCatAt(text, it) else null) else - (if (it > 0) text[it - 1] else null); + (if (it > 0) hxCatAt(text, prevGrapheme(text, it)) else null); // skip any initial newline characters while (true) { - const ch: u8 = if (is_prev) blk: { - if (it == 0) break; - break :blk text[it - 1]; - } else blk: { - if (it >= text.len) break; - break :blk text[it]; - }; - if (ch != '\n' and ch != '\r') break; - if (is_prev) it -= 1 else it += 1; - prev_ch = ch; - if (is_prev) head -|= 1 else head += 1; + if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break; + const cell = if (is_prev) prevGrapheme(text, it) else it; + const cat = hxCatAt(text, cell); + if (cat != .eol) break; + it = if (is_prev) cell else nextGrapheme(text, cell); + prev_cat = cat; + head = it; } - if (prev_ch != null and hxCat(prev_ch.?) == .eol) anchor = head; + if (prev_cat == .eol) anchor = head; // find the target position const head_start = head; while (true) { - const next_ch: u8 = if (is_prev) blk: { - if (it == 0) break; - it -= 1; - break :blk text[it]; - } else blk: { - if (it >= text.len) break; - const c = text[it]; - it += 1; - break :blk c; - }; - if (prev_ch == null or hxReached(target, prev_ch.?, next_ch)) { + if ((is_prev and it == 0) or (!is_prev and it >= text.len)) break; + const cell = if (is_prev) prevGrapheme(text, it) else it; + const next_cat = hxCatAt(text, cell); + if (prev_cat == null or hxReached(target, prev_cat.?, next_cat)) { if (head == head_start) anchor = head else break; } - prev_ch = next_ch; - if (is_prev) head -|= 1 else head += 1; + prev_cat = next_cat; + it = if (is_prev) cell else nextGrapheme(text, cell); + head = it; } return .{ .anchor = anchor, .head = head }; } @@ -1007,31 +1117,31 @@ pub fn hxVertTarget(text: []const u8, pos: usize, down: bool, count: usize, goal const s = lineStartOffset(text, nline); // hxLineEndIdx(text, nline) without its second walk to nline (see hxOff) const e = std.mem.indexOfScalarPos(u8, text, s, '\n') orelse text.len; - return @min(s + goal_col, e); + return s + graphemeStart(text[s..e], @min(goal_col, e - s)); } /// f/F/t/T target cell. helix find_char: the exclusive (till) search starts /// one further out so repeats make progress; not-found = null (no move). -pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bool, count: usize) ?usize { +pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u21, fwd: bool, till: bool, count: usize) ?usize { var left = @max(1, count); if (fwd) { const head = nextGrapheme(text, cursor); - var i = if (till) head + 1 else head; + var i = if (till) nextGrapheme(text, head) else head; if (i > text.len) return null; - while (i < text.len) : (i += 1) { - if (text[i] == ch) { + while (i < text.len) : (i = nextGrapheme(text, i)) { + if (codepointAt(text, i) == ch) { left -= 1; - if (left == 0) return if (till) i - 1 else i; + if (left == 0) return if (till) prevGrapheme(text, i) else i; } } return null; } - var i = if (till) cursor -| 1 else cursor; + var i = if (till) prevGrapheme(text, cursor) else cursor; while (i > 0) { - i -= 1; - if (text[i] == ch) { + i = prevGrapheme(text, i); + if (codepointAt(text, i) == ch) { left -= 1; - if (left == 0) return if (till) i + 1 else i; + if (left == 0) return if (till) nextGrapheme(text, i) else i; } } return null; @@ -1040,26 +1150,19 @@ pub fn hxFindTarget(text: []const u8, cursor: usize, ch: u8, fwd: bool, till: bo // helix textobject.rs find_word_boundary fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usize { var prev: HxCat = if (fwd) - (if (pos0 == 0) .ws else hxCat(text[pos0 - 1])) + (if (pos0 == 0) .ws else hxCatAt(text, prevGrapheme(text, pos0))) else - (if (pos0 >= text.len) .ws else hxCat(text[pos0])); + (if (pos0 >= text.len) .ws else hxCatAt(text, pos0)); var pos = pos0; var it = pos0; while (true) { - const ch: u8 = if (fwd) blk: { - if (it >= text.len) break; - const c = text[it]; - it += 1; - break :blk c; - } else blk: { - if (it == 0) break; - it -= 1; - break :blk text[it]; - }; - const cat = hxCat(ch); + if ((fwd and it >= text.len) or (!fwd and it == 0)) break; + const cell = if (fwd) it else prevGrapheme(text, it); + const cat = hxCatAt(text, cell); if (cat == .eol or cat == .ws) return pos; if (!long and cat != prev and pos != 0 and pos != text.len) return pos; - if (fwd) pos += 1 else pos -|= 1; + it = if (fwd) nextGrapheme(text, cell) else cell; + pos = it; prev = cat; } return pos; @@ -1070,14 +1173,20 @@ fn hxFindWordBoundary(text: []const u8, pos0: usize, fwd: bool, long: bool) usiz pub fn hxTextobjectWord(text: []const u8, r: HxRange, around: bool, long: bool) HxRange { const pos = hxCursor(text, r); const word_start = hxFindWordBoundary(text, pos, false, long); - const cat: HxCat = if (pos < text.len) hxCat(text[pos]) else .ws; - const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, pos + 1, true, long); + const cat: HxCat = if (pos < text.len) hxCatAt(text, pos) else .ws; + const word_end = if (cat == .ws or cat == .eol) pos else hxFindWordBoundary(text, nextGrapheme(text, pos), true, long); if (word_start == word_end or !around) return .{ .anchor = word_start, .head = word_end }; var end = word_end; - while (end < text.len and hxIsWs(text[end]) and hxCat(text[end]) != .eol) end += 1; + while (end < text.len and hxIsWs(hxCatAt(text, end)) and hxCatAt(text, end) != .eol) + end = nextGrapheme(text, end); if (end > word_end) return .{ .anchor = word_start, .head = end }; var start = word_start; - while (start > 0 and hxIsWs(text[start - 1]) and hxCat(text[start - 1]) != .eol) start -= 1; + while (start > 0) { + const before = prevGrapheme(text, start); + const before_cat = hxCatAt(text, before); + if (!hxIsWs(before_cat) or before_cat == .eol) break; + start = before; + } return .{ .anchor = start, .head = word_end }; } @@ -1294,6 +1403,38 @@ test "hx find targets" { try std.testing.expectEqual(@as(?usize, null), hxFindTarget(t, 0, 'z', true, false, 1)); } +test "extended grapheme boundaries cover combining emoji flag and CJK text" { + const text = "a" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "🇧🇷" ++ "界"; + const boundaries = [_]usize{ 0, 1, 4, 19, 27, 30 }; + for (boundaries[0 .. boundaries.len - 1], boundaries[1..]) |start, end| { + try std.testing.expectEqual(end, nextGrapheme(text, start)); + try std.testing.expectEqual(start, prevGrapheme(text, end)); + } + // Stale byte offsets are repaired to a cluster boundary instead of being + // allowed to leak continuation bytes into cursor state. + try std.testing.expectEqual(@as(usize, 1), graphemeStart(text, 2)); + try std.testing.expectEqual(@as(usize, 1), prevGrapheme(text, 3)); + try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3)); +} + +test "Unicode find and word motion stay on grapheme boundaries" { + const lines = [_][]const u8{"\u{e9}x\u{e9}"}; + try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'é', true, false, 1).?); + + const text = "café 世界 ok\n"; + const first = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 1, .next_word_start); + try std.testing.expectEqual(@as(usize, 0), first.anchor); + try std.testing.expectEqual(@as(usize, 6), first.head); + const second = hxWordMove(text, .{ .anchor = 0, .head = 1 }, 2, .next_word_start); + try std.testing.expectEqual(@as(usize, 6), second.anchor); + try std.testing.expectEqual(@as(usize, 13), second.head); + + // Long-word motions split on Unicode whitespace, not only ASCII spaces. + const nbsp = "alpha\u{a0}beta\n"; + const long = hxWordMove(nbsp, .{ .anchor = 0, .head = 1 }, 1, .next_long_word_start); + try std.testing.expectEqual(@as(usize, 7), long.head); +} + test "hx increment" { const a = std.testing.allocator; { @@ -1335,7 +1476,7 @@ test "kindOf" { test "char/line motions" { const lines = [_][]const u8{ "alpha beta", " two words", "x" }; const c = Cursor{ .row = 0, .col = 5 }; - try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(c)); + try std.testing.expectEqual(Cursor{ .row = 0, .col = 4 }, charLeft(&.{"hello"}, c)); try std.testing.expectEqual(Cursor{ .row = 0, .col = 6 }, charRight(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 1, .col = 5 }, lineDown(&lines, c)); try std.testing.expectEqual(Cursor{ .row = 0, .col = 5 }, lineUp(&lines, Cursor{ .row = 1, .col = 5 })); @@ -1440,6 +1581,19 @@ test "deleteChar" { try std.testing.expectEqualStrings("abc", r2); } +test "Unicode edits replace and delete whole graphemes" { + const a = std.testing.allocator; + const content = "A" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "🇧🇷" ++ "界" ++ "Z"; + + const deleted = try deleteChar(a, content, .{ .row = 0, .col = 2 }); + defer a.free(deleted); + try std.testing.expectEqualStrings("A👩🏽\u{200d}🚀🇧🇷界Z", deleted); + + const replaced = try replaceChars(a, content, .{ .row = 0, .col = 1 }, .{ .row = 0, .col = 19 }, '界'); + defer a.free(replaced); + try std.testing.expectEqualStrings("A界界界界Z", replaced); +} + test "deleteLines middle" { const content = "one\ntwo\nthree\nfour"; const a = std.testing.allocator; diff --git a/src/normal_input.zig b/src/normal_input.zig index c6b2574f..4b047a7f 100644 --- a/src/normal_input.zig +++ b/src/normal_input.zig @@ -263,7 +263,7 @@ pub const Action = union(enum) { goto: struct { target: Goto, count: u32, explicit_count: bool }, view: View, find: struct { kind: Find, char: u21, count: u32 }, - replace_char: u8, + replace_char: u21, match_bracket, textobject: struct { char: u21, around: bool }, surround_add: u21, @@ -409,8 +409,7 @@ pub fn parse(state: *State, key: Input) Result { .replace => { state.prefix = .none; const char = key.literal() orelse return .ignored; - if (char > 0x7f) return .ignored; - return resultAction(.{ .replace_char = @intCast(char) }); + return resultAction(.{ .replace_char = char }); }, .match => { if (state.match_sub == .none) { @@ -615,6 +614,13 @@ test "modified and special keys cannot satisfy literal continuations" { try std.testing.expectEqual(State{}, state); } +test "replace accepts a Unicode literal" { + var state: State = .{}; + try std.testing.expectEqual(Result.pending, parse(&state, input('r', &.{.prefix_replace}))); + try std.testing.expectEqualDeep(Result{ .action = .{ .replace_char = '界' } }, parse(&state, input('界', &.{}))); + try std.testing.expectEqual(State{}, state); +} + test "once versus per-selection is semantic action metadata" { try std.testing.expectEqual(Scope.once, (@as(Action, .search)).scope()); try std.testing.expectEqual(Scope.once, (Action{ .edit = .{ .kind = .undo, .count = 1 } }).scope()); diff --git a/src/pardes.zig b/src/pardes.zig index 38e317c4..5bf03788 100644 --- a/src/pardes.zig +++ b/src/pardes.zig @@ -24,6 +24,7 @@ pub const animation = @import("animation.zig"); pub const panel_animation = @import("panel_animation.zig"); const ghostty_vt = @import("ghostty-vt"); const uucode = @import("uucode"); +const vaxis = @import("vaxis"); const mvzr = @import("mvzr"); const modal = @import("modal.zig"); const normal_input = @import("normal_input.zig"); @@ -1554,7 +1555,7 @@ const WordBounds = struct { lo: usize, hi: usize }; /// The whitespace-delimited word covering `col` in `str`. The topbar uses the /// same bounds for pointer feedback and dispatch, so the word that lights up /// is necessarily the word a middle click will execute. -fn wordBoundsAtCol(str: []const u8, col: u16) ?WordBounds { +fn wordBoundsAtCol(str: []const u8, col: usize) ?WordBounds { if (col >= str.len or str[col] == ' ') return null; var lo: usize = col; while (lo > 0 and str[lo - 1] != ' ') lo -= 1; @@ -1564,7 +1565,7 @@ fn wordBoundsAtCol(str: []const u8, col: u16) ?WordBounds { } /// the whitespace-delimited word covering `col` in `str` (topbar dispatch) -fn wordAtCol(str: []const u8, col: u16) []const u8 { +fn wordAtCol(str: []const u8, col: usize) []const u8 { const bounds = wordBoundsAtCol(str, col) orelse return ""; return str[bounds.lo..bounds.hi]; } @@ -2118,8 +2119,26 @@ pub const Surface = struct { (std.unicode.utf8Decode(text[i .. i + n]) catch null) else null; - var cp_slice = if (decoded == null) "\u{FFFD}" else text[i .. i + n]; - i += if (decoded == null) 1 else n; + var cp_slice: []const u8 = "\u{FFFD}"; + var consumed: usize = 1; + if (decoded != null) { + // A surface cell is a grapheme, not a codepoint. Keeping the + // complete cluster makes combining marks visible and keeps ZWJ, + // modifier, flag and Indic sequences in the same screen cell + // that cursor/edit math treats as one unit. + var git = uucode.grapheme.utf8Iterator(text[i..]); + if (git.nextGrapheme()) |g| { + const candidate = text[i .. i + g.end]; + if (std.unicode.utf8ValidateSlice(candidate)) { + cp_slice = candidate; + consumed = candidate.len; + } else { + cp_slice = text[i .. i + n]; + consumed = n; + } + } + } + i += consumed; var cp = decoded orelse 0xFFFD; if (cp == '\r') continue; // A Surface cell is already positioned, not a terminal byte @@ -2137,8 +2156,10 @@ pub const Surface = struct { cp = 0xFFFD; cp_slice = "\u{FFFD}"; } - const width: u16 = if (cp < 0x80) 1 else uucode.get(.width, cp); - if (width == 0) continue; + const width: u16 = if (decoded == null or (cp < 0x80 and cp_slice.len == 1)) + 1 + else + @max(1, vaxis.gwidth.gwidth(cp_slice, .unicode)); // a DOUBLE-width glyph with one column left is not drawn at all. // Writing it puts one cell in the surface and two on the glass, and // when that column is the screen's last the terminal wraps the tail @@ -2148,7 +2169,21 @@ pub const Surface = struct { // problem. Reachable from any byte cut through wide text: hscroll's // and soft wrap's both. if (width == 2 and col + 1 >= end) break; - s.set(col, y, cp_slice, style); + // The cross-host Cell ABI has seven payload bytes. Preserve a valid + // codepoint prefix when a modern emoji cluster is longer; its full + // width and edit boundary still come from the complete cluster. + var shown = cp_slice; + if (shown.len > @typeInfo(@FieldType(Cell, "text")).array.len) { + const cap = @typeInfo(@FieldType(Cell, "text")).array.len; + var prefix: usize = 0; + while (prefix < shown.len) { + const cp_len = std.unicode.utf8ByteSequenceLength(shown[prefix]) catch break; + if (prefix + cp_len > cap) break; + prefix += cp_len; + } + shown = if (prefix > 0) shown[0..prefix] else "\u{FFFD}"; + } + s.set(col, y, shown, style); if (width == 2 and col + 1 < end) { // spacer: empty cell under the wide glyph's tail s.set(col + 1, y, "", style); @@ -2199,6 +2234,106 @@ test "surface print expands configured tabs and normalizes other controls" { try std.testing.expectEqualStrings("A", cells[cell_count - 1].grapheme()); } +test "surface print keeps combining and wide graphemes in their display cells" { + var cells: [8]Cell = @splat(.{}); + var surface = Surface{ .cols = cells.len, .rows = 1, .cells = &cells }; + + const end = surface.print(0, 0, cells.len, "e\u{301}界👩🏽\u{200d}🚀1\u{fe0f}\u{20e3}A", .{}); + try std.testing.expectEqual(@as(u16, 8), end); + try std.testing.expectEqualStrings("e\u{301}", cells[0].grapheme()); + try std.testing.expectEqualStrings("界", cells[1].grapheme()); + try std.testing.expectEqualStrings("", cells[2].grapheme()); + try std.testing.expectEqualStrings("👩", cells[3].grapheme()); + try std.testing.expectEqualStrings("", cells[4].grapheme()); + try std.testing.expectEqualStrings("1\u{fe0f}\u{20e3}", cells[5].grapheme()); + try std.testing.expectEqualStrings("", cells[6].grapheme()); + try std.testing.expectEqualStrings("A", cells[7].grapheme()); +} + +test "insert and normal modes edit complete Unicode graphemes" { + const gpa = std.testing.allocator; + const p = try Pardes.init(gpa, .{ .cols = 60, .rows = 12 }); + defer p.deinit(); + const pane = try p.hxOpenFileContent("a" ++ "e\u{301}" ++ "👩🏽\u{200d}🚀" ++ "界" ++ "z"); + + pane.mode = .insert; + pane.cur_col = 22; // on z, after the CJK grapheme + p.insertKey(pane, .{ .cp = Key.left }); + try std.testing.expectEqual(@as(i32, 19), pane.cur_col); + p.insertKey(pane, .{ .cp = Key.left }); + try std.testing.expectEqual(@as(i32, 4), pane.cur_col); + p.insertKey(pane, .{ .cp = Key.right }); + try std.testing.expectEqual(@as(i32, 19), pane.cur_col); + p.insertKey(pane, .{ .cp = Key.backspace }); + try std.testing.expectEqualStrings("ae\u{301}界z", pane.file.?.content); + try std.testing.expectEqual(@as(i32, 4), pane.cur_col); + + pane.mode = .normal; + pane.cur_col = 1; + p.normalDelete(pane, false); + try std.testing.expectEqualStrings("a界z", pane.file.?.content); + try std.testing.expectEqual(@as(i32, 1), pane.cur_col); + p.normalReplaceChar(pane, 'λ'); + try std.testing.expectEqualStrings("aλz", pane.file.?.content); + + file_pane.setContent(p, &pane.file.?, try gpa.dupe(u8, "界a")); + pane.cur_col = 3; + pane.vsel = .{ .active = true, .row = 0, .col = 0, .explicit = true }; + p.normalReplaceChar(pane, 'λ'); + try std.testing.expectEqualStrings("λλ", pane.file.?.content); + try std.testing.expectEqual(@as(i32, 2), pane.cur_col); + try std.testing.expectEqual(@as(i32, 0), pane.vsel.col); + + file_pane.setContent(p, &pane.file.?, try gpa.dupe(u8, "界a")); + pane.cur_col = 0; + pane.vsel = .{ .active = true, .row = 0, .col = 3, .explicit = true }; + p.normalReplaceChar(pane, 'λ'); + try std.testing.expectEqualStrings("λλ", pane.file.?.content); + try std.testing.expectEqual(@as(i32, 0), pane.cur_col); + try std.testing.expectEqual(@as(i32, 2), pane.vsel.col); +} + +test "Unicode display cells map back to body and tag byte cursors" { + const gpa = std.testing.allocator; + const p = try Pardes.init(gpa, .{ .cols = 60, .rows = 12 }); + defer p.deinit(); + const pane = try p.hxOpenFileContent("a界e\u{301}z\n"); + const f = &pane.file.?; + p.gpa.free(f.path); + f.path = try p.gpa.dupe(u8, "界e\u{301}.txt"); + + var frame = std.heap.ArenaAllocator.init(gpa); + defer frame.deinit(); + _ = try p.render(frame.allocator()); + const rect = p.rects[0]; + const text_x = rect.x + config.GUTTER + config.PREFIX_W; + const body_y = rect.y + BOX_H; + // The second screen cell is the trailing half of the wide CJK grapheme. + // Both halves map to its one byte boundary. + p.update(.{ .mouse = .{ .button = config.select_button, .kind = .press, .col = text_x + 2, .row = body_y } }); + p.update(.{ .mouse = .{ .button = config.select_button, .kind = .release, .col = text_x + 2, .row = body_y } }); + try std.testing.expectEqual(@as(i32, 1), pane.cur_col); + + // A click in the wide path glyph likewise becomes a byte cursor at the + // grapheme start; arrow motion then advances by the full UTF-8 cluster. + p.enterTagEdit(pane, 1); + try std.testing.expectEqual(@as(u16, 0), pane.tag_col); + p.tagInsertKey(pane, .{ .cp = Key.right }); + try std.testing.expectEqual(@as(u16, 3), pane.tag_col); + p.tagInsertKey(pane, .{ .cp = Key.right }); + try std.testing.expectEqual(@as(u16, 6), pane.tag_col); + + const before = try gpa.dupe(u8, pane.tagSlice()); + defer gpa.free(before); + p.enterTagEdit(pane, -1); + const insertion = pane.tag_col; + p.tagInsertKey(pane, .{ .cp = 'λ', .text = "λ" }); + try std.testing.expectEqual(insertion + 2, pane.tag_col); + p.tagInsertKey(pane, .{ .cp = Key.backspace }); + try std.testing.expectEqual(insertion, pane.tag_col); + try std.testing.expectEqualStrings(before, pane.tagSlice()); +} + test "tabbed file aligns syntax cursor and mouse while preserving virtual columns" { const gpa = std.testing.allocator; const p = try Pardes.init(gpa, .{ .cols = 60, .rows = 12 }); @@ -2920,12 +3055,13 @@ pub const Pane = struct { /// the body mode a tag edit hijacked (tags are always insert); terminals /// restore it on exit so clicking the tag never changes the pane's mode tag_mode: Mode = .normal, - /// THE tag coordinate space: byte columns into the WHOLE rendered tag, - /// prefix ++ tail (tagText). One space, no conversions — the renderer, the - /// mouse and every motion speak it directly. The prefix is live chrome, so - /// it is selectable/yankable/executable but READ-ONLY: every edit op - /// measures from `edit0` (= tagPrefix().len, the first editable column) and - /// does nothing left of it. + /// THE tag coordinate space: UTF-8 byte offsets into the WHOLE rendered + /// tag, prefix ++ tail (tagText), always on grapheme boundaries. Motions + /// and edits use these offsets; rendering and pointer input convert at the + /// screen boundary. The prefix is live chrome, so it is selectable, + /// yankable and executable but READ-ONLY: every edit op measures from + /// `edit0` (= tagPrefix().len, the first editable byte) and does nothing + /// left of it. tag_col: u16 = 0, tag_anchor: u16 = 0, ed_undo: [term_pane.history_max]term_pane.Snapshot = undefined, @@ -4540,7 +4676,7 @@ pub const Pardes = struct { leader_on: bool = false, leader_keys: [4]u8 = undefined, leader_n: u8 = 0, - /// the TOPBAR holds the keyboard, parked at this column of the rendered + /// the TOPBAR holds the keyboard, parked at this UTF-8 byte offset of the /// row-0 line. Global like leader_on for the same reason: row 0 is not a /// pane and never will be, so its one piece of focus state cannot live on /// one. `null` = the panes have the keyboard, which is every other frame. @@ -5401,12 +5537,9 @@ pub const Pardes = struct { /// the row below kept its out at the pad. Now the column moves out /// together, so the words stay in a line and a click walks down them. /// - /// Bytes, not display columns — like every other tag coordinate (see - /// Pane.tag_col). A multibyte path renders narrower than it measures, so - /// the line is straight for ASCII paths and drifts by a column per wide - /// character otherwise. That is the same approximation the plain - /// right-align always made; fixing it is a job for the whole tag - /// coordinate space, not for this scan. + /// Alignment is measured in display cells. Cursor and selection state use + /// UTF-8 byte offsets, then map through the same grapheme-width helpers at + /// the screen boundary. /// /// Two rules make up the "when possible": /// @@ -5450,7 +5583,7 @@ pub const Pardes = struct { pane_tail; const laid = if (q.tag_init) q.tagSlice() else words; const lead = laid.len - std.mem.trimStart(u8, laid, " ").len; - const q_end = (p.tagPrefix(q) catch continue).len + lead + std.mem.trimStart(u8, words, " ").len; + const q_end = file_pane.displayWidth(p.tagPrefix(q) catch continue) + lead + file_pane.displayWidth(std.mem.trimStart(u8, words, " ")); end = @max(end, @min(q_end, tw)); }; return end -| used; @@ -5462,7 +5595,7 @@ pub const Pardes = struct { fn tagText(p: *Pardes, arena: std.mem.Allocator, pane: *Pane) ![]u8 { const prefix = try p.tagPrefix(pane); const tail = curTail(pane); - const gap = p.tagGap(pane, prefix.len + tail.len); + const gap = p.tagGap(pane, file_pane.displayWidth(prefix) + file_pane.displayWidth(tail)); const out = try arena.alloc(u8, prefix.len + gap + tail.len); @memcpy(out[0..prefix.len], prefix); @memset(out[prefix.len..][0..gap], ' '); @@ -5476,8 +5609,8 @@ pub const Pardes = struct { fn seedTail(p: *Pardes, pane: *Pane) void { if (pane.tag_init) return; const tail = curTail(pane); - const prefix_len = (p.tagPrefix(pane) catch return).len; - const gap = p.tagGap(pane, prefix_len + tail.len); + const prefix = (p.tagPrefix(pane) catch return); + const gap = p.tagGap(pane, file_pane.displayWidth(prefix) + file_pane.displayWidth(tail)); if (gap + tail.len > pane.tag_tail.len) return; @memset(pane.tag_tail[0..gap], ' '); @memcpy(pane.tag_tail[gap..][0..tail.len], tail); @@ -5486,7 +5619,7 @@ pub const Pardes = struct { } /// focus the tag for editing, seeding the tail on first touch and parking - /// the cursor at `col` — a column of the RENDERED tag (see tag_col). + /// the byte cursor at the grapheme displayed under screen column `col`. /// NEGATIVE means the tail's first WORD, which is where `:` and a tagline /// hop land: a place no click can name, so it needs no sentinel of its own /// and the callers need no prefix length. @@ -5509,7 +5642,12 @@ pub const Pardes = struct { // the door. The spaces stay editable; h and Left still walk into them. const tail = pane.tagSlice(); const lead: i32 = @intCast(tail.len - std.mem.trimStart(u8, tail, " ").len); - pane.tag_col = @intCast(if (col < 0) @min(edit0 + lead, end) else std.math.clamp(col, 0, end)); + if (col < 0) { + pane.tag_col = @intCast(@min(edit0 + lead, end)); + } else { + const text = p.tagText(p.scratch.allocator(), pane) catch return; + pane.tag_col = @intCast(@min(text.len, file_pane.rawAtDisplay(text, @intCast(col)))); + } } fn exitTagEdit(pane: *Pane) void { @@ -5538,7 +5676,7 @@ pub const Pardes = struct { const text = p.tagText(p.scratch.allocator(), pane) catch return null; if (pane.tag_sel) { const b = tagSelBounds(pane); - const hi = @min(b.hi + 1, text.len); + const hi = modal.nextGrapheme(text, b.hi); return if (hi > b.lo) text[b.lo..hi] else null; } const b = config.wordBounds(text, @min(@as(usize, pane.tag_col), text.len)); @@ -5580,14 +5718,19 @@ pub const Pardes = struct { } switch (key.cp) { Key.backspace => if (pane.tag_col > edit0) { - pane.removeTagByte(pane.tag_col - edit0 - 1); - pane.tag_col -= 1; + const text = p.tagText(p.scratch.allocator(), pane) catch return; + const prev = @max(@as(usize, edit0), modal.prevGrapheme(text, pane.tag_col)); + const count = @as(usize, pane.tag_col) - prev; + for (0..count) |_| pane.removeTagByte(prev - edit0); + pane.tag_col = @intCast(prev); }, Key.left => if (pane.tag_col > 0) { - pane.tag_col -= 1; + const text = p.tagText(p.scratch.allocator(), pane) catch return; + pane.tag_col = @intCast(modal.prevGrapheme(text, pane.tag_col)); }, Key.right => if (pane.tag_col < end) { - pane.tag_col += 1; + const text = p.tagText(p.scratch.allocator(), pane) catch return; + pane.tag_col = @intCast(modal.nextGrapheme(text, pane.tag_col)); }, Key.home => pane.tag_col = 0, Key.end => pane.tag_col = end, @@ -5601,8 +5744,8 @@ pub const Pardes = struct { const lo = @min(r.anchor, r.head); const hi = @max(r.anchor, r.head); pane.tag_col = @intCast(modal.hxCursor(text, r)); - pane.tag_sel = hi > lo + 1; // a 1-wide range IS the block cursor - if (pane.tag_sel) pane.tag_anchor = @intCast(if (r.head > r.anchor) lo else hi - 1); + pane.tag_sel = modal.nextGrapheme(text, lo) < hi; // one grapheme IS the block cursor + if (pane.tag_sel) pane.tag_anchor = @intCast(if (r.head > r.anchor) lo else modal.prevGrapheme(text, hi)); } /// normal mode ON the tag — where `:` lands. The body's own helix motions @@ -5757,7 +5900,7 @@ pub const Pardes = struct { // on it does. Leave the bar FIRST — `Kill` lives up here and tears the // session down, the same hazard the pane-tag chord has with `Del`. if (hit(key, config.look_key) or hit(key, config.exec_key)) { - const word = wordAtCol(bar, @intCast(cur)); + const word = wordAtCol(bar, cur); p.topbar_col = null; if (word.len > 0) _ = p.execute(p.active, word); return; @@ -5819,7 +5962,7 @@ pub const Pardes = struct { while (count_it.next()) |line| : (count_row += 1) { if (count_row < r0 or count_row > r1) continue; const b0 = @min(file_pane.renderedLineByteCol(pane, count_row, line, c0), line.len); - const b1 = @min(file_pane.renderedLineByteCol(pane, count_row, line, c1) + 1, line.len); + const b1 = modal.nextGrapheme(line, @min(file_pane.renderedLineByteCol(pane, count_row, line, c1), line.len)); total += b1 - b0 + @intFromBool(selected > 0); selected += 1; } @@ -5836,7 +5979,7 @@ pub const Pardes = struct { } first = false; const b0 = @min(file_pane.renderedLineByteCol(pane, v, line, c0), line.len); - const b1 = @min(file_pane.renderedLineByteCol(pane, v, line, c1) + 1, line.len); + const b1 = modal.nextGrapheme(line, @min(file_pane.renderedLineByteCol(pane, v, line, c1), line.len)); @memcpy(out[at..][0 .. b1 - b0], line[b0..b1]); at += b1 - b0; } @@ -5891,10 +6034,17 @@ pub const Pardes = struct { /// the word under the modal cursor as a pane-local selection (paneText /// coords: row 0 is the tag; file panes carry the line-number prefix) - fn cursorWordSel(pane: *Pane) Sel { + fn cursorWordSel(p: *Pardes, pane: *Pane) Sel { const w = pane.wrapRow(pane.cur_row, pane.cur_col); const vrow = w.row + @as(i32, BOX_H); - const vcol = if (pane.file != null) file_pane.displayOffset(pane, pane.cur_row, w.at, pane.cur_col) + @as(i32, config.PREFIX_W) else pane.cur_col; + const vcol = if (pane.file != null) + file_pane.displayOffset(pane, pane.cur_row, w.at, pane.cur_col) + @as(i32, config.PREFIX_W) + else blk: { + const pl = p.paneCursorLines(pane) catch break :blk pane.cur_col; + const local = pane.cur_row - pl.row0; + if (local < 0 or @as(usize, @intCast(local)) >= pl.lines.len) break :blk pane.cur_col; + break :blk file_pane.lineDisplayOffset(pl.lines[@intCast(local)], @intCast(@max(0, w.at)), @intCast(@max(0, pane.cur_col))); + }; return .{ .state = .done, .c0 = vcol, .c1 = vcol, .r0 = vrow, .r1 = vrow }; } @@ -5930,8 +6080,13 @@ pub const Pardes = struct { const visible = clicked.r0 - @as(i32, BOX_H); const wrapped = pane.wrapAt(visible); const row = wrapped.line; - const col = if (pane.file != null) - file_pane.byteAtRowDisplay(pane, wrapped.line, wrapped.at, clicked.c0 - @as(i32, config.PREFIX_W)) + const col = if (clicked.r0 >= BOX_H) + p.paneByteAtDisplay( + pane, + wrapped.line, + wrapped.at, + clicked.c0 - (if (pane.file != null) @as(i32, config.PREFIX_W) else 0), + ) else clicked.c0; var result: PointerOperand = .{ .row = row, .col = col }; @@ -6157,7 +6312,7 @@ pub const Pardes = struct { // to the file-ish word under the cursor. if (pane.mode == .normal and (hit(key, config.look_key) or hit(key, config.exec_key))) { const cmd = if (hit(key, config.look_key)) config.look_cmd else config.exec_cmd; - pane.pinCursor(); + p.pinPaneCursor(pane); const explicit = (p.native_images and hasPdfSelection(pane)) or (pane.vsel.active and pane.vsel.explicit) or pane.msel.active; if (explicit) { @@ -6169,7 +6324,7 @@ pub const Pardes = struct { return; } } - const sel = p.expandedSel(pane, cursorWordSel(pane)) orelse return; + const sel = p.expandedSel(pane, p.cursorWordSel(pane)) orelse return; const word = p.selectionText(pane, sel) catch return; p.runBuiltin(cmd, p.active, "", word); return; @@ -6439,6 +6594,29 @@ pub const Pardes = struct { return .{ .lines = try term_pane.cursorLines(p, pane), .row0 = 0 }; } + fn paneByteAtDisplay(p: *Pardes, pane: *Pane, row: i32, from_raw: i32, display_col: i32) i32 { + if (pane.file != null) + return file_pane.byteAtRowDisplay(pane, row, from_raw, display_col); + const pl = p.paneCursorLines(pane) catch return @max(0, from_raw + display_col); + const local = row - pl.row0; + if (local < 0 or @as(usize, @intCast(local)) >= pl.lines.len) + return @max(0, from_raw + display_col); + return @intCast(file_pane.byteAtDisplayFrom( + pl.lines[@intCast(local)], + @intCast(@max(0, from_raw)), + @intCast(@max(0, display_col)), + )); + } + + /// Freeze the live terminal cursor into the modal coordinate space. The + /// emulator reports screen cells; editing state stores UTF-8 byte offsets. + fn pinPaneCursor(p: *Pardes, pane: *Pane) void { + if (pane.cur_pinned) return; + pane.pinCursor(); + if (pane.file == null and !hasPdf(pane)) + pane.cur_col = p.paneByteAtDisplay(pane, pane.cur_row, 0, pane.cur_col); + } + fn toModalCursor(pane: *Pane, pl: PaneLines) modal.Cursor { const r: i32 = pane.cur_row - pl.row0; return .{ .row = @intCast(@max(0, r)), .col = @intCast(@max(0, pane.cur_col)) }; @@ -6450,6 +6628,18 @@ pub const Pardes = struct { pane.cur_pinned = true; } + fn insertVerticalCursor(lines: []const []const u8, c: modal.Cursor, down: bool) modal.Cursor { + if (lines.len == 0) return c; + const row = if (down) @min(c.row + 1, lines.len - 1) else c.row -| 1; + const target = lines[row]; + if (target.len == 0) return .{ .row = row, .col = 0 }; + const source = if (c.row < lines.len) lines[c.row] else ""; + const goal = file_pane.rawDisplayCol(source, c.col); + const mapped = file_pane.rawAtDisplay(target, goal); + const last = modal.prevGrapheme(target, target.len); + return .{ .row = row, .col = modal.graphemeStart(target, @min(mapped, last)) }; + } + // ---- helix range plumbing (see modal.zig "helix range engine") ---- // The pane's cursor + vsel cells render ONE helix gap range over the flat // motion surface. Every motion builds the current range, transforms it the @@ -7222,9 +7412,8 @@ pub const Pardes = struct { /// f/t/F/T: anchor at the old cursor cell, head on the hit (not found: no move) fn findMove(pane: *Pane, pl: PaneLines, text: []const u8, range: modal.HxRange, ch: u21, fwd: bool, till: bool, cnt: usize) void { - if (ch > 0x7f) return; // ponytail: ASCII targets only (byte columns) const cur = modal.hxCursor(text, range); - const t = modal.hxFindTarget(text, cur, @intCast(ch), fwd, till, cnt) orelse return; + const t = modal.hxFindTarget(text, cur, ch, fwd, till, cnt) orelse return; const res = if (pane.select) modal.hxPutCursor(text, range, t, true) else @@ -7236,13 +7425,17 @@ pub const Pardes = struct { fn verticalMove(pane: *Pane, pl: PaneLines, text: []const u8, range: modal.HxRange, down: bool, cnt: usize) void { const cur = modal.hxCursor(text, range); const pos = panePos(pane, text, cur); - const goal: usize = if (pane.sticky_col >= 0) @intCast(pane.sticky_col) else pos.col; + const goal: usize = if (pane.sticky_col >= 0) + @intCast(pane.sticky_col) + else + file_pane.rawDisplayCol(modal.lineSlice(text, pos.row), pos.col); // modal.hxVertTarget with the row we already have and the indexed // offset conversion — it would otherwise recount the buffer's newlines // and walk to the target line, two more full passes per j/k const last_row = paneLineCount(pane, text) - 1; const nline = if (down) @min(pos.row + @max(1, cnt), last_row) else pos.row -| @max(1, cnt); - const t = paneOff(pane, text, .{ .row = nline, .col = goal }); + const target_col = file_pane.rawAtDisplay(modal.lineSlice(text, nline), goal); + const t = paneOff(pane, text, .{ .row = nline, .col = target_col }); // extend mode never walks onto the empty trailing line (helix) if (pane.select and t == text.len and text.len > 0 and text[text.len - 1] == '\n') return; setPaneRange(pane, pl, text, modal.hxPutCursor(text, range, t, pane.select), false); @@ -7373,7 +7566,7 @@ pub const Pardes = struct { /// complete before this function runs; this switch reads document state /// only to execute the already-recognized action. fn executeNormalAction(p: *Pardes, pane: *Pane, semantic: normal_input.Action) void { - pane.pinCursor(); + p.pinPaneCursor(pane); const pl = p.paneCursorLines(pane) catch return; const text = p.flatSurface(pane, pl) catch return; const lines = pl.lines; @@ -7410,7 +7603,8 @@ pub const Pardes = struct { .column => { const line = modal.hxLineOf(text, cur); const ls = modal.lineStartOffset(text, line); - return pointMove(pane, pl, text, range, @min(ls + (@as(usize, go.count) - 1), modal.hxLineEndIdx(text, line))); + const slice = text[ls..modal.hxLineEndIdx(text, line)]; + return pointMove(pane, pl, text, range, ls + modal.graphemeAtColumn(slice, @as(usize, go.count) - 1)); }, .view_top => return gotoWindow(pane, pl, text, range, .top, go.count), .view_center => return gotoWindow(pane, pl, text, range, .center, go.count), @@ -7506,14 +7700,14 @@ pub const Pardes = struct { .next_long_word_end => return wordMove(pane, pl, text, range, move.count, .next_long_word_end), }, .repeat_find => |count| { - if (pane.find_op == 0 or pane.find_ch > 0x7f) return; + if (pane.find_op == 0) return; const fwd = pane.find_op == config.find_char_fwd or pane.find_op == config.till_char_fwd; const till = pane.find_op == config.till_char_fwd or pane.find_op == config.till_char_back; var repeated = range; var moved = false; for (0..count) |_| { const cc = modal.hxCursor(text, repeated); - const target = modal.hxFindTarget(text, cc, @intCast(pane.find_ch), fwd, till, 1) orelse break; + const target = modal.hxFindTarget(text, cc, pane.find_ch, fwd, till, 1) orelse break; repeated = if (pane.select) modal.hxPutCursor(text, repeated, target, true) else @@ -7815,7 +8009,7 @@ pub const Pardes = struct { pane.tag_sel = false; pane.mode = .insert; pane.pending = 0; - // tag_col is a rendered-tag column; the prompt offset slices tag_tail. + // tag_col and the prompt offset are both UTF-8 byte offsets. pane.tag_col = @intCast((p.tagPrefix(pane) catch return).len + pane.tag_tail_len); } @@ -8531,7 +8725,7 @@ pub const Pardes = struct { /// cursor on its FIRST — the same shape the old terminal stepper left, and /// the reason `col0` is the position the walk compares against. fn landLookSpot(p: *Pardes, id: usize, pane: *Pane, spot: LookSpot) void { - pane.pinCursor(); // fresh out of tty mode the cursor still tracks the shell + p.pinPaneCursor(pane); // fresh out of tty mode the cursor still tracks the shell pane.vsel = .{ .active = true, .row = spot.row, .col = spot.col1, .explicit = true }; pane.msel.active = false; pane.nsel = 0; @@ -8572,7 +8766,7 @@ pub const Pardes = struct { fn enterInsert(p: *Pardes, pane: *Pane, where: InsertAt, cnt: usize) void { if (hasPdf(pane)) return; - pane.pinCursor(); + p.pinPaneCursor(pane); // snapshot once per insert session (WITH the pre-insert selection) so // `u` undoes the whole session and restores what was selected p.pushUndo(pane); @@ -8694,7 +8888,7 @@ pub const Pardes = struct { // its only pardes use (the acme chords) needs explicit selections // anyway, and those never enter insert mode pane.vsel.active = false; - if (!pane.cur_pinned) pane.pinCursor(); + if (!pane.cur_pinned) p.pinPaneCursor(pane); // arrows and paging are pure motion over the WHOLE surface, so they // run before editText — a terminal must not freeze shell rows into an // edit buffer just because you walked across them @@ -8703,10 +8897,10 @@ pub const Pardes = struct { const pl = p.paneCursorLines(pane) catch return; const cur0 = toModalCursor(pane, pl); const nc = switch (key.cp) { - Key.left => modal.charLeft(cur0), + Key.left => modal.charLeft(pl.lines, cur0), Key.right => modal.charRight(pl.lines, cur0), - Key.up => modal.lineUp(pl.lines, cur0), - Key.down => modal.lineDown(pl.lines, cur0), + Key.up => insertVerticalCursor(pl.lines, cur0, false), + Key.down => insertVerticalCursor(pl.lines, cur0, true), else => cur0, }; fromModalCursor(pane, pl, nc); @@ -8829,9 +9023,11 @@ pub const Pardes = struct { }, Key.backspace => { if (pane.cur_col > 0) { - const new = modal.deleteChar(p.gpa, text, .{ .row = c.row, .col = c.col - 1 }) catch return; + const line = modal.lineSlice(text, c.row); + const prev = modal.prevGrapheme(line, c.col); + const new = modal.deleteChar(p.gpa, text, .{ .row = c.row, .col = prev }) catch return; p.setEditText(pane, new); - pane.cur_col -= 1; + pane.cur_col = @intCast(prev); pane.cur_pinned = true; pane.ensureCursorVisible(); return; @@ -9115,17 +9311,19 @@ pub const Pardes = struct { .{ .row = @intCast(@max(0, b.lo_row - row0)), .col = @intCast(@max(0, b.lo_col)) } else blk: { const hrow: usize = @intCast(@max(0, b.hi_row - row0)); - const gcol = @as(usize, @intCast(@max(0, b.hi_col))) + 1; - if (gcol > modal.lineSlice(eb.text, hrow).len) break :blk .{ .row = hrow + 1, .col = 0 }; + const hi_col: usize = @intCast(@max(0, b.hi_col)); + const high_line = modal.lineSlice(eb.text, hrow); + if (hi_col >= high_line.len) break :blk .{ .row = hrow + 1, .col = 0 }; + const gcol = modal.nextGrapheme(high_line, hi_col); break :blk .{ .row = hrow, .col = gcol }; }; const out = modal.insertAt(p.gpa, eb.text, at, y) catch return; p.setEditText(pane, out); - pane.vsel = .{ .active = y.len > 1, .row = @as(i32, @intCast(at.row)) + row0, .col = @intCast(at.col), .explicit = false }; + pane.vsel = .{ .active = modal.nextGrapheme(y, 0) < y.len, .row = @as(i32, @intCast(at.row)) + row0, .col = @intCast(at.col), .explicit = false }; const end = modal.advanceBy(at, y); if (end.col > 0) { pane.cur_row = @as(i32, @intCast(end.row)) + row0; - pane.cur_col = @intCast(end.col - 1); + pane.cur_col = @intCast(modal.prevGrapheme(modal.lineSlice(out, end.row), end.col)); } else { pane.cur_row = @as(i32, @intCast(end.row -| 1)) + row0; pane.cur_col = @intCast(modal.lineSlice(out, end.row -| 1).len); @@ -9261,7 +9459,7 @@ pub const Pardes = struct { const r0: usize = @intCast(@max(0, @min(pane.msel.r0, pane.msel.r1) - row0)); const r1: usize = @intCast(@max(0, @max(pane.msel.r0, pane.msel.r1) - row0)); const llen = modal.lineSlice(content, r1).len; - return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = llen -| 1 } }; + return .{ .a = .{ .row = r0, .col = 0 }, .b = .{ .row = r1, .col = if (llen == 0) 0 else modal.prevGrapheme(modal.lineSlice(content, r1), llen) } }; } const c = modal.Cursor{ .row = @intCast(@max(0, pane.cur_row - row0)), .col = @intCast(@max(0, pane.cur_col)) }; return .{ .a = c, .b = c }; @@ -9281,13 +9479,27 @@ pub const Pardes = struct { /// `r<ch>`: overwrite the selection (or the cursor char) with ch — /// newlines included (helix), so `xrz` joins the selected lines - fn normalReplaceChar(p: *Pardes, pane: *Pane, ch: u8) void { + fn normalReplaceChar(p: *Pardes, pane: *Pane, ch: u21) void { pane.select = false; const eb = p.editTextEol(pane, selRows(pane)) orelse return; + const before = paneRange(pane, eb.text, eb.row0); + const lo = @min(before.anchor, before.head); + const hi = @max(before.anchor, before.head); + var graphemes: usize = 0; + var at = lo; + while (at < hi) : (graphemes += 1) at = modal.nextGrapheme(eb.text, at); + var encoded: [4]u8 = undefined; + const encoded_len = std.unicode.utf8Encode(ch, &encoded) catch return; const r = selRange(pane, eb.text, eb.row0); p.pushUndo(pane); const new = modal.replaceChars(p.gpa, eb.text, r.a, r.b, ch) catch return; p.setEditText(pane, new); + const end = lo + graphemes * encoded_len; + const mapped: modal.HxRange = if (before.head < before.anchor) + .{ .anchor = end, .head = lo } + else + .{ .anchor = lo, .head = end }; + setPaneRange(pane, .{ .lines = &.{}, .row0 = eb.row0 }, new, mapped, pane.vsel.explicit); } /// `R`: replace the selection (or the cursor char) with the DEFAULT @@ -9307,11 +9519,11 @@ pub const Pardes = struct { const new = modal.replaceRange(p.gpa, eb.text, r.a, r.b, y) catch return; p.setEditText(pane, new); pane.msel.active = false; - pane.vsel = .{ .active = y.len > 1, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false }; + pane.vsel = .{ .active = modal.nextGrapheme(y, 0) < y.len, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false }; const end = modal.advanceBy(r.a, y); if (end.col > 0) { pane.cur_row = @as(i32, @intCast(end.row)) + eb.row0; - pane.cur_col = @intCast(end.col - 1); + pane.cur_col = @intCast(modal.prevGrapheme(modal.lineSlice(new, end.row), end.col)); } else { // the yank ended in '\n': the cursor lands ON that newline pane.cur_row = @as(i32, @intCast(end.row -| 1)) + eb.row0; @@ -9637,8 +9849,8 @@ pub const Pardes = struct { const new = modal.replaceRange(p.gpa, eb.text, r.a, r.b, rep) catch return; p.setEditText(pane, new); const start = modal.lineStartOffset(new, r.a.row) + r.a.col; - const cc = modal.hxPos(new, start + rep.len - 1); - pane.vsel = .{ .active = rep.len > 1, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false }; + const cc = modal.hxPos(new, modal.prevGrapheme(new, start + rep.len)); + pane.vsel = .{ .active = modal.nextGrapheme(rep, 0) < rep.len, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false }; pane.msel.active = false; pane.cur_row = @as(i32, @intCast(cc.row)) + eb.row0; pane.cur_col = @intCast(cc.col); @@ -9728,14 +9940,15 @@ pub const Pardes = struct { const eb = p.editTextEol(pane, selRows(pane)) orelse return; const r = selRange(pane, eb.text, eb.row0); p.pushUndo(pane); - var new = modal.insertAt(p.gpa, eb.text, .{ .row = r.b.row, .col = r.b.col + 1 }, &[1]u8{pr.c}) catch return; + const close_col = modal.nextGrapheme(modal.lineSlice(eb.text, r.b.row), r.b.col); + var new = modal.insertAt(p.gpa, eb.text, .{ .row = r.b.row, .col = close_col }, &[1]u8{pr.c}) catch return; p.setEditText(pane, new); new = modal.insertAt(p.gpa, new, .{ .row = r.a.row, .col = r.a.col }, &[1]u8{pr.o}) catch return; p.setEditText(pane, new); pane.msel.active = false; pane.vsel = .{ .active = true, .row = @as(i32, @intCast(r.a.row)) + eb.row0, .col = @intCast(r.a.col), .explicit = false }; pane.cur_row = @as(i32, @intCast(r.b.row)) + eb.row0; - pane.cur_col = @intCast(r.b.col + 1 + @as(usize, if (r.a.row == r.b.row) 1 else 0)); + pane.cur_col = @intCast(close_col + @as(usize, if (r.a.row == r.b.row) 1 else 0)); pane.cur_pinned = true; pane.sticky_col = -1; pane.ensureCursorVisible(); @@ -10329,7 +10542,13 @@ pub const Pardes = struct { // next cursor move or left wheel pulls it back } else if (pane.file != null and !p.settings.wrap) { // wrapped there is nothing off to the right to reach - pane.hscroll = @max(0, pane.hscroll + (if (m.button == .wheel_right) config.wheel_cols else -config.wheel_cols)); + const line = file_pane.sourceLine(pane, pane.cur_row); + const visual = file_pane.rawDisplayCol(line, @intCast(@max(0, pane.hscroll))); + const next: usize = if (m.button == .wheel_right) + visual +| @as(usize, @intCast(config.wheel_cols)) + else + visual -| @as(usize, @intCast(config.wheel_cols)); + pane.hscroll = @intCast(file_pane.rawAtDisplay(line, next)); } }, config.select_button => switch (m.kind) { @@ -10474,7 +10693,8 @@ pub const Pardes = struct { // click (a stray select click must never Kill) if (m.button == config.exec_button) { var tb_buf: [1200]u8 = undefined; - const word = wordAtCol(p.topbar(&tb_buf), mcol); + const bar = p.topbar(&tb_buf); + const word = wordAtCol(bar, file_pane.rawAtDisplay(bar, mcol)); // a topbar word runs on the PRESS — there is no // drag to chord into, so the argument is simply // whatever is selected right now: select a word, @@ -10615,9 +10835,9 @@ pub const Pardes = struct { // row is ignored (a tag is one line) and the anchor stays where // the press put it, so this is the mouse's `v` const pane = p.panes[d.id] orelse return; - const end: i32 = @intCast((p.tagPrefix(pane) catch return).len + pane.tag_tail_len); const c = @as(i32, mcol) - @as(i32, p.rects[d.id].x + config.GUTTER); - pane.tag_col = @intCast(std.math.clamp(c, 0, end)); + const text = p.tagText(p.scratch.allocator(), pane) catch return; + pane.tag_col = @intCast(@min(text.len, file_pane.rawAtDisplay(text, @intCast(@max(0, c))))); pane.tag_sel = pane.tag_col != pane.tag_anchor; }, .none => {}, @@ -10722,10 +10942,12 @@ pub const Pardes = struct { // pointer whether or not that row is a continuation const w = pane.wrapAt(body_vis); pane.cur_row = w.line; - pane.cur_col = if (pane.file != null) - file_pane.byteAtRowDisplay(pane, w.line, w.at, sl.c1 - @as(i32, config.PREFIX_W)) - else - sl.c1; + pane.cur_col = p.paneByteAtDisplay( + pane, + w.line, + w.at, + sl.c1 - (if (pane.file != null) @as(i32, config.PREFIX_W) else 0), + ); pane.cur_pinned = true; if (!pane.isTerminal()) pane.mode = .normal; pane.msel.active = false; @@ -10874,8 +11096,8 @@ pub const Pardes = struct { const w1 = pane.wrapAt(@max(0, sl.r1 - @as(i32, BOX_H))); const row0 = w0.line; const row1 = w1.line; - const col0 = @max(0, sl.c0 - pfx) + w0.at; - const col1 = @max(0, sl.c1 - pfx) + w1.at; + const col0 = p.paneByteAtDisplay(pane, w0.line, w0.at, sl.c0 - pfx); + const col1 = p.paneByteAtDisplay(pane, w1.line, w1.at, sl.c1 - pfx); pane.cur_row = row1; pane.cur_col = col1; pane.cur_pinned = true; @@ -10898,11 +11120,11 @@ pub const Pardes = struct { p.pushUndo(pane); const new = modal.insertAt(p.gpa, f.content, at, y) catch return; file_pane.setContent(p, f, new); - pane.vsel = .{ .active = y.len > 1, .row = @intCast(at.row), .col = @intCast(at.col), .explicit = false }; + pane.vsel = .{ .active = modal.nextGrapheme(y, 0) < y.len, .row = @intCast(at.row), .col = @intCast(at.col), .explicit = false }; const end = modal.advanceBy(at, y); if (end.col > 0) { pane.cur_row = @intCast(end.row); - pane.cur_col = @intCast(end.col - 1); + pane.cur_col = @intCast(modal.prevGrapheme(modal.lineSlice(f.content, end.row), end.col)); } else { // the register ended in '\n': the cursor lands ON that newline pane.cur_row = @intCast(end.row -| 1); @@ -11192,6 +11414,14 @@ pub const Pardes = struct { if (comptime !pdf_enabled) return; if (pane.pdf == null) return; var state = paneNormalState(pane); + // A PDF has no body character to find, so its bare `f` is the direct + // document-outline door. Prefix continuations still go through the + // shared parser, and text/terminal panes retain `f<char>` unchanged. + if (state.prefix == .none and isPrefix(key, 'f')) { + state.clear(); + putPaneNormalState(pane, state); + return p.runBuiltin(.PdfSections, p.active, "", null); + } const parsed = normal_input.parse(&state, normalInput(key)); putPaneNormalState(pane, state); switch (parsed) { @@ -12584,17 +12814,20 @@ pub const Pardes = struct { const w: u16 = @intCast(iw); if (w < tw) _ = s.print(tx + tw - w, row, w, ibuf[0..iw], msg_style); } - // ...and the cursor follows the text it edits. tag_col is a - // rendered-tag column, so the prompt's own column is it minus where - // the marker starts; a cursor LEFT of that is still over the part + // ...and the cursor follows the text it edits. tag_col is a byte + // offset, so the prompt maps its suffix through display widths; a + // cursor LEFT of the marker is still over the part // of the tag that stayed on the tagline, and the tag cursor // renderPane already placed there is the right one. if (id != p.active) continue; const at = prompt_at orelse continue; const prompt0 = (p.tagPrefix(pane) catch continue).len + at; const col = @as(usize, pane.tag_col); - if (col >= prompt0 and col - prompt0 < tw) - s.cursor = .{ .x = tx + @as(u16, @intCast(col - prompt0)), .y = row, .bar = pane.mode == .insert }; + if (col >= prompt0) { + const prompt_col = file_pane.displayWidth(text[0..@min(col - prompt0, text.len)]); + if (prompt_col < tw) + s.cursor = .{ .x = tx + @as(u16, @intCast(prompt_col)), .y = row, .bar = pane.mode == .insert }; + } } // global tagbar: full width, top row @@ -12617,9 +12850,11 @@ pub const Pardes = struct { // feedback at all. Paint the same word the click dispatcher resolves, // immediately, while leaving whitespace inert. if (p.pointer_inside and p.hover_row < TOPBAR_H) { - if (wordBoundsAtCol(p.topbar(&tb_buf), p.hover_col)) |bounds| { - var col: usize = bounds.lo; - while (col < bounds.hi and col < s.cols) : (col += 1) { + const bar = p.topbar(&tb_buf); + if (wordBoundsAtCol(bar, file_pane.rawAtDisplay(bar, p.hover_col))) |bounds| { + var col = file_pane.rawDisplayCol(bar, bounds.lo); + const hi = file_pane.rawDisplayCol(bar, bounds.hi); + while (col < hi and col < s.cols) : (col += 1) { const cell = s.at(@intCast(col), 0); cell.default = false; cell.style.bg = .{ .rgb = th.sel_bg }; @@ -12630,9 +12865,10 @@ pub const Pardes = struct { // the topbar's cursor, if it has the keyboard. AFTER the pane loop on // purpose: there is exactly one Surface cursor and the bar's must beat // the active pane's. Always a block — the bar has no insert mode. - if (p.topbar_col) |c| if (c < s.cols) { - s.cursor = .{ .x = c, .y = 0, .bar = false }; - }; + if (p.topbar_col) |c| { + const col = file_pane.rawDisplayCol(p.topbar(&tb_buf), c); + if (col < s.cols) s.cursor = .{ .x = @intCast(col), .y = 0, .bar = false }; + } // resize-handle hint / drag previews: a dash overlay that keeps the // underlying colors (border drags + hover), or the move indicator. A @@ -13046,7 +13282,7 @@ pub const Pardes = struct { }; } // tag char selection highlight (helix v/x, or a tagline sweep), - // inclusive [lo, hi] — rendered-tag columns, so no conversion. + // inclusive [lo, hi], mapped from byte offsets to display cells. // // Every selection on screen paints in the LIVE theme (`th`) and not in // `chrome`: a highlight is not attached to any geometry, it appears @@ -13055,8 +13291,9 @@ pub const Pardes = struct { // animation's frames. if (pane.tag_edit and pane.tag_sel) { const b = tagSelBounds(pane); - var col: usize = b.lo; - const end: usize = b.hi; + var col = file_pane.rawDisplayCol(tag, b.lo); + const hi = modal.nextGrapheme(tag, b.hi); + const end = file_pane.rawDisplayCol(tag, hi) -| 1; while (col <= end and col < tw) : (col += 1) { const cell = s.at(tx + @as(u16, @intCast(col)), tag_y); cell.default = false; @@ -13064,10 +13301,11 @@ pub const Pardes = struct { cell.style.fg = .{ .rgb = th.sel_fg }; } } - // cursor while editing the tag: a rendered-tag column, verbatim + // cursor while editing the tag: byte offset mapped to its display cell if (active and pane.tag_edit) { // bar while typing, block for `:` normal mode (same rule as a body) - if (pane.tag_col < tw) s.cursor = .{ .x = tx + pane.tag_col, .y = tag_y, .bar = pane.mode == .insert }; + const col = file_pane.rawDisplayCol(tag, pane.tag_col); + if (col < tw) s.cursor = .{ .x = tx + @as(u16, @intCast(col)), .y = tag_y, .bar = pane.mode == .insert }; } // A native PDF page uses the same backend-neutral pixel attachment as @@ -13243,12 +13481,19 @@ pub const Pardes = struct { while (vr + @as(i32, BOX_H) < @as(i32, r.h)) : (vr += 1) { const w = pane.wrapAt(vr); if (w.line < bnd.lo_row or w.line > bnd.hi_row) continue; + const visible_line = modal.lineSlice(body, @intCast(vr)); const cstart: i32 = if (w.line == bnd.lo_row) - (if (pane.file != null) file_pane.displayOffset(pane, w.line, w.at, bnd.lo_col) else bnd.lo_col - w.at) + vpfx + (if (pane.file != null) + file_pane.displayOffset(pane, w.line, w.at, bnd.lo_col) + else + file_pane.lineDisplayOffset(visible_line, @intCast(@max(0, w.at)), @intCast(@max(0, bnd.lo_col)))) + vpfx else vpfx; const cend: i32 = if (w.line == bnd.hi_row) - (if (pane.file != null) file_pane.displayEndOffset(pane, w.line, w.at, bnd.hi_col) else bnd.hi_col - w.at) + vpfx + (if (pane.file != null) + file_pane.displayEndOffset(pane, w.line, w.at, bnd.hi_col) + else + file_pane.lineDisplayEndOffset(visible_line, @intCast(@max(0, w.at)), @intCast(@max(0, bnd.hi_col)))) + vpfx else @as(i32, tw) - 1; var col: i32 = @max(cstart, vpfx); @@ -13263,7 +13508,14 @@ pub const Pardes = struct { if (hover_only) continue; // quiet preview preserves the source ink const cw = pane.wrapRow(sr.row, sr.col); const crow = cw.row + @as(i32, BOX_H); - const ccol = (if (pane.file != null) file_pane.displayOffset(pane, sr.row, cw.at, sr.col) else sr.col - cw.at) + vpfx; + const ccol = (if (pane.file != null) + file_pane.displayOffset(pane, sr.row, cw.at, sr.col) + else + file_pane.lineDisplayOffset( + modal.lineSlice(body, @intCast(@max(0, cw.row))), + @intCast(@max(0, cw.at)), + @intCast(@max(0, sr.col)), + )) + vpfx; if (crow >= BOX_H and crow < @as(i32, r.h) and ccol >= vpfx and ccol < tw) { const cell = s.at(tx + @as(u16, @intCast(ccol)), body_y + @as(u16, @intCast(crow - BOX_H))); cell.default = false; @@ -13290,6 +13542,12 @@ pub const Pardes = struct { // cells, so account for every expanded tab before the cursor. const cx = if (pane.file != null) @as(i32, config.PREFIX_W) + file_pane.displayOffset(pane, crow, cwp.at, ccol) + else if (pane.cur_pinned) + file_pane.lineDisplayOffset( + modal.lineSlice(body, @intCast(@max(0, cwp.row))), + @intCast(@max(0, cwp.at)), + @intCast(@max(0, ccol)), + ) else ccol; if (prow >= BOX_H and cx >= 0 and prow < r.h and cx < tw) diff --git a/src/pdf_pane_integration_test.zig b/src/pdf_pane_integration_test.zig index eaebae5e..b290616e 100644 --- a/src/pdf_pane_integration_test.zig +++ b/src/pdf_pane_integration_test.zig @@ -111,7 +111,8 @@ test "PdfSections Look follows the exact owning PDF, not an equal path" { const p = try Pardes.init(gpa, .{ .file = path, .cols = 80, .rows = 28 }); defer p.deinit(); const first = p.panes[0].?; - pdf_pane.openSections(p, 0); + // Bare normal-mode `f` on a PDF dispatches the PdfSections builtin. + p.update(.{ .key = .{ .cp = 'f', .text = "f" } }); const first_output_id = first.search_pane orelse return error.MissingPdfSectionsOutput; const first_output = p.panes[first_output_id].?; try std.testing.expect(pdf_pane.isSectionsOutput(first_output)); diff --git a/src/syntax.zig b/src/syntax.zig index b7b9f32f..def11420 100644 --- a/src/syntax.zig +++ b/src/syntax.zig @@ -1,7 +1,7 @@ //! Tree-sitter syntax highlighting: one style byte per content byte, filled by //! running each grammar's highlights.scm query (slurped at build time into the //! ts_queries options module). Grammar set is tiered: `zig`; `minimal` (c, cpp, -//! zig); and `full`, which adds ~23 languages lazily on first use. +//! zig); and `full`, which adds ~24 languages lazily on first use. const std = @import("std"); const config = @import("pardes_config"); const tracy = @import("tracy.zig"); @@ -173,6 +173,7 @@ fn synFor(name: []const u8) Syn { .{ "string", .string }, .{ "character", .string }, .{ "number", .number }, + .{ "numeric", .number }, .{ "float", .number }, .{ "boolean", .number }, .{ "keyword", .keyword }, @@ -239,3 +240,26 @@ test "tree-sitter allocator callbacks preserve and free exact allocations" { try std.testing.expect(syntaxRealloc(live, 0) == null); live = null; } + +test "default full grammar set highlights Typst source" { + if (!enabled or !full_grammars) return; + start(std.testing.allocator); + defer stop(); + + const source = "// note\n#let answer = 42\n#let text = \"hello\"\n"; + const styles = try highlightFileRange(std.testing.allocator, "paper.typst", source, 0, source.len); + defer std.testing.allocator.free(styles); + + const comment_at = std.mem.indexOf(u8, source, "// note").?; + const keyword_at = std.mem.indexOf(u8, source, "let").?; + const number_at = std.mem.indexOf(u8, source, "42").?; + const string_at = std.mem.indexOf(u8, source, "\"hello\"").?; + try std.testing.expectEqual(Syn.comment, @as(Syn, @enumFromInt(styles[comment_at]))); + try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(styles[keyword_at]))); + try std.testing.expectEqual(Syn.number, @as(Syn, @enumFromInt(styles[number_at]))); + try std.testing.expectEqual(Syn.string, @as(Syn, @enumFromInt(styles[string_at]))); + + const short_ext = try highlightFileRange(std.testing.allocator, "paper.typ", source, 0, source.len); + defer std.testing.allocator.free(short_ext); + try std.testing.expectEqual(Syn.keyword, @as(Syn, @enumFromInt(short_ext[keyword_at]))); +} diff --git a/test/snapshots/badutf.golden b/test/snapshots/badutf.golden index dceef522..fda9f8a7 100644 --- a/test/snapshots/badutf.golden +++ b/test/snapshots/badutf.golden @@ -32,7 +32,7 @@ == snap hscroll-midglyph grid=50x30 cursor=46,2 |New Newcol Find Grep Help Tutor Dump NextColor Deb | /tmp/pardes-snap/badutf/cwd/wide.txt Save New De -| 1 ��語 日本語 日本語 日本語 tail +| 1 日本語 日本語 日本語 日本語 日本語 tail | 2 | 3 | diff --git a/test/snapshots/badutf.snap b/test/snapshots/badutf.snap index 49d90bdd..46592178 100644 --- a/test/snapshots/badutf.snap +++ b/test/snapshots/badutf.snap @@ -1,7 +1,7 @@ # Invalid UTF-8 must RENDER, never panic. Two sources, both ordinary use: # a file pane holds whatever bytes are on disk (latin-1, a truncated sequence, -# an ELF opened by mistake), and the byte-column hscroll cuts a multi-byte -# glyph in half by design. Undecodable bytes come out as U+FFFD, one per byte. +# an ELF opened by mistake). Undecodable bytes come out as U+FFFD, while +# horizontal scrolling must start at a complete grapheme boundary. file bad.txt hello \xff\xfe world\nlatin-1 caf\xe9 tail\ntruncated \xe6\x97 here\nend\n file wide.txt \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e \xe6\x97\xa5\xe6\x9c\xac\xe8\xaa\x9e tail\nsecond line\n start 30 100 bad.txt @@ -9,15 +9,14 @@ wait 8000 New Newcol stable 700 20000 snap badbytes # 50 columns, not 100: a FILE boots alone and fills the window, and at 100 the -# whole line fits, so `$` scrolls nowhere and there is no glyph to cut. This is +# whole line fits, so `$` scrolls nowhere. This is # the width the left column used to be, i.e. the pane this case was written for start 30 50 wide.txt wait 8000 New Newcol stable 700 20000 # SPC t w turns soft wrap OFF: hscroll is the other half of this case and a -# wrapped body has nothing to scroll sideways, so the byte cut that halves a -# glyph only exists here. (The WRAP cut halves one too — same bytes, same -# U+FFFD — and wrap.snap pins that one.) +# wrapped body has nothing to scroll sideways. The unwrapped view must keep the +# first visible CJK glyph intact instead of starting inside its UTF-8 bytes. key space t w stable 400 5000 key $ |
