diff options
Diffstat (limited to 'src/file_pane.zig')
| -rw-r--r-- | src/file_pane.zig | 97 |
1 files changed, 89 insertions, 8 deletions
diff --git a/src/file_pane.zig b/src/file_pane.zig index e0150de2..5c7f4b77 100644 --- a/src/file_pane.zig +++ b/src/file_pane.zig @@ -16,14 +16,9 @@ const syntax = @import("syntax.zig"); const tracy = @import("tracy.zig"); const term_pane = @import("term_pane.zig"); const dump = @import("dump.zig"); +const limits = @import("limits.zig"); const SYNTAX_CONTEXT_AFTER_ROWS: usize = 2; -/// EDIT BOUNDARIES REMEMBERED PER FILE PANE. Every entry owns a gpa copy of -/// the WHOLE file, so this number multiplies heap, not just the pane: 256 of -/// them is not a bound a 384 KiB board could ever reach anyway. `pushHistory` -/// evicts and frees the oldest once full, so the smaller ring loses the -/// deepest undo steps and nothing else — no truncation, no dropped edit. -const undo_max = if (@import("pardes_config").platform == .p4) 16 else 256; /// Content and primary selection at one file edit boundary. Keeping only the /// primary avoids putting pardes.MAX_SELS ranges in every history entry. @@ -54,9 +49,9 @@ pub const State = struct { highlights: []u8 = &.{}, highlight_start: usize = 0, syntax_dirty: bool = true, - undo: [undo_max]Snapshot = undefined, + undo: [limits.undo_max]Snapshot = undefined, undo_len: usize = 0, - redo: [undo_max]Snapshot = undefined, + redo: [limits.undo_max]Snapshot = undefined, redo_len: usize = 0, }; @@ -220,6 +215,92 @@ test "display columns map complete Unicode graphemes" { try std.testing.expectEqual(@as(usize, 2), graphemeDisplayWidth("👩\u{200d}🚀")); } +test "the ASCII arm of graphemeDisplayWidth matches the gwidth it skips" { + // The arm claims a one-byte printable ASCII grapheme is one cell without asking `gwidth`. That + // is only worth having if the two never disagree, so ask both for every byte the arm can see - + // including \t, \r, the rest of the C0 controls and DEL, which the range test excludes and + // which must therefore still come back from `gwidth` (or, for the tab, from the config). + const ref = struct { + fn width(grapheme: []const u8) usize { + if (std.mem.eql(u8, grapheme, "\t")) return config.tab_width; + return @max(1, @as(usize, vaxis.gwidth.gwidth(grapheme, .unicode))); + } + }.width; + + var one: [1]u8 = undefined; + var b: u8 = 0; + while (b < 0x80) : (b += 1) { + one[0] = b; + try std.testing.expectEqual(ref(one[0..1]), graphemeDisplayWidth(one[0..1])); + } + // Multi-byte clusters never reach the arm (len != 1), so they pin that it does not widen its + // claim: a combining sequence and a ZWJ emoji are one and two cells, a CJK glyph is two, and + // an invalid byte is the one cell `gwidth` reports for U+FFFD-shaped input. + for ([_][]const u8{ + "e\u{301}", "a\u{903}", "1\u{fe0f}\u{20e3}", "\u{4e16}", + "\u{1f642}", "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8", + }) |g| try std.testing.expectEqual(ref(g), graphemeDisplayWidth(g)); +} + +test "the ASCII run in fitEnd survives an exhaustive byte sweep" { + // The case list in the test above is hand-picked; this one is not. Every byte 0x00..0x7f is + // placed next to every neighbour that can change the answer - a combining mark, a ZWJ + // sequence, a spacing mark, a variation selector, a wide glyph, and a bad start byte, a + // truncated tail and a bad continuation - and every break column is compared against the + // grapheme walk. An off-by-one column here moves text between wrapped rows, so equality is + // exact, not approximate. + const reference = struct { + fn fitEnd(text: []const u8, start: usize, width: usize) usize { + var end = start; + var used: usize = 0; + while (end < text.len) { + const next_end = modal.nextGrapheme(text, end); + const next_used = used +| graphemeDisplayWidth(text[end..next_end]); + if (next_used > width) return if (end == start) next_end else end; + used = next_used; + end = next_end; + } + return end; + } + }.fitEnd; + + const neighbours = [_][]const u8{ + "", "z", "\u{301}", "\u{200d}\u{1f680}", + "\u{903}", "\u{fe0f}", "\u{4e16}", "\u{1f642}", + "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8", "\xe4\x28\xb8", + }; + var buf: [16]u8 = undefined; + // The same pair again behind an ASCII prefix, so a break can land exactly at the run boundary + // as well as before it and inside the multi-byte cluster that follows it. + var prefixed: [18]u8 = undefined; + prefixed[0] = 'a'; + prefixed[1] = 'b'; + var b: u8 = 0; + while (b < 0x80) : (b += 1) { + buf[0] = b; + for (neighbours) |tail| { + @memcpy(buf[1..][0..tail.len], tail); + const pair = buf[0 .. 1 + tail.len]; + @memcpy(prefixed[2..][0..pair.len], pair); + for ([_][]const u8{ pair, prefixed[0 .. 2 + pair.len] }) |text| { + var width: usize = 0; + while (width <= text.len + 3) : (width += 1) { + var start: usize = 0; + while (start <= text.len) : (start += 1) { + std.testing.expectEqual( + reference(text, start, width), + fitEnd(text, start, width), + ) catch |e| { + std.debug.print("fitEnd({any}, {d}, {d})\n", .{ text, start, width }); + return e; + }; + } + } + } + } + } +} + fn fitEnd(text: []const u8, start: usize, width: usize) usize { var end = start; var used: usize = 0; |
