summaryrefslogtreecommitdiff
path: root/src/file_pane.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/file_pane.zig')
-rw-r--r--src/file_pane.zig97
1 files changed, 89 insertions, 8 deletions
diff --git a/src/file_pane.zig b/src/file_pane.zig
index e0150de2..5c7f4b77 100644
--- a/src/file_pane.zig
+++ b/src/file_pane.zig
@@ -16,14 +16,9 @@ const syntax = @import("syntax.zig");
const tracy = @import("tracy.zig");
const term_pane = @import("term_pane.zig");
const dump = @import("dump.zig");
+const limits = @import("limits.zig");
const SYNTAX_CONTEXT_AFTER_ROWS: usize = 2;
-/// EDIT BOUNDARIES REMEMBERED PER FILE PANE. Every entry owns a gpa copy of
-/// the WHOLE file, so this number multiplies heap, not just the pane: 256 of
-/// them is not a bound a 384 KiB board could ever reach anyway. `pushHistory`
-/// evicts and frees the oldest once full, so the smaller ring loses the
-/// deepest undo steps and nothing else — no truncation, no dropped edit.
-const undo_max = if (@import("pardes_config").platform == .p4) 16 else 256;
/// Content and primary selection at one file edit boundary. Keeping only the
/// primary avoids putting pardes.MAX_SELS ranges in every history entry.
@@ -54,9 +49,9 @@ pub const State = struct {
highlights: []u8 = &.{},
highlight_start: usize = 0,
syntax_dirty: bool = true,
- undo: [undo_max]Snapshot = undefined,
+ undo: [limits.undo_max]Snapshot = undefined,
undo_len: usize = 0,
- redo: [undo_max]Snapshot = undefined,
+ redo: [limits.undo_max]Snapshot = undefined,
redo_len: usize = 0,
};
@@ -220,6 +215,92 @@ test "display columns map complete Unicode graphemes" {
try std.testing.expectEqual(@as(usize, 2), graphemeDisplayWidth("👩\u{200d}🚀"));
}
+test "the ASCII arm of graphemeDisplayWidth matches the gwidth it skips" {
+ // The arm claims a one-byte printable ASCII grapheme is one cell without asking `gwidth`. That
+ // is only worth having if the two never disagree, so ask both for every byte the arm can see -
+ // including \t, \r, the rest of the C0 controls and DEL, which the range test excludes and
+ // which must therefore still come back from `gwidth` (or, for the tab, from the config).
+ const ref = struct {
+ fn width(grapheme: []const u8) usize {
+ if (std.mem.eql(u8, grapheme, "\t")) return config.tab_width;
+ return @max(1, @as(usize, vaxis.gwidth.gwidth(grapheme, .unicode)));
+ }
+ }.width;
+
+ var one: [1]u8 = undefined;
+ var b: u8 = 0;
+ while (b < 0x80) : (b += 1) {
+ one[0] = b;
+ try std.testing.expectEqual(ref(one[0..1]), graphemeDisplayWidth(one[0..1]));
+ }
+ // Multi-byte clusters never reach the arm (len != 1), so they pin that it does not widen its
+ // claim: a combining sequence and a ZWJ emoji are one and two cells, a CJK glyph is two, and
+ // an invalid byte is the one cell `gwidth` reports for U+FFFD-shaped input.
+ for ([_][]const u8{
+ "e\u{301}", "a\u{903}", "1\u{fe0f}\u{20e3}", "\u{4e16}",
+ "\u{1f642}", "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8",
+ }) |g| try std.testing.expectEqual(ref(g), graphemeDisplayWidth(g));
+}
+
+test "the ASCII run in fitEnd survives an exhaustive byte sweep" {
+ // The case list in the test above is hand-picked; this one is not. Every byte 0x00..0x7f is
+ // placed next to every neighbour that can change the answer - a combining mark, a ZWJ
+ // sequence, a spacing mark, a variation selector, a wide glyph, and a bad start byte, a
+ // truncated tail and a bad continuation - and every break column is compared against the
+ // grapheme walk. An off-by-one column here moves text between wrapped rows, so equality is
+ // exact, not approximate.
+ const reference = struct {
+ fn fitEnd(text: []const u8, start: usize, width: usize) usize {
+ var end = start;
+ var used: usize = 0;
+ while (end < text.len) {
+ const next_end = modal.nextGrapheme(text, end);
+ const next_used = used +| graphemeDisplayWidth(text[end..next_end]);
+ if (next_used > width) return if (end == start) next_end else end;
+ used = next_used;
+ end = next_end;
+ }
+ return end;
+ }
+ }.fitEnd;
+
+ const neighbours = [_][]const u8{
+ "", "z", "\u{301}", "\u{200d}\u{1f680}",
+ "\u{903}", "\u{fe0f}", "\u{4e16}", "\u{1f642}",
+ "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8", "\xe4\x28\xb8",
+ };
+ var buf: [16]u8 = undefined;
+ // The same pair again behind an ASCII prefix, so a break can land exactly at the run boundary
+ // as well as before it and inside the multi-byte cluster that follows it.
+ var prefixed: [18]u8 = undefined;
+ prefixed[0] = 'a';
+ prefixed[1] = 'b';
+ var b: u8 = 0;
+ while (b < 0x80) : (b += 1) {
+ buf[0] = b;
+ for (neighbours) |tail| {
+ @memcpy(buf[1..][0..tail.len], tail);
+ const pair = buf[0 .. 1 + tail.len];
+ @memcpy(prefixed[2..][0..pair.len], pair);
+ for ([_][]const u8{ pair, prefixed[0 .. 2 + pair.len] }) |text| {
+ var width: usize = 0;
+ while (width <= text.len + 3) : (width += 1) {
+ var start: usize = 0;
+ while (start <= text.len) : (start += 1) {
+ std.testing.expectEqual(
+ reference(text, start, width),
+ fitEnd(text, start, width),
+ ) catch |e| {
+ std.debug.print("fitEnd({any}, {d}, {d})\n", .{ text, start, width });
+ return e;
+ };
+ }
+ }
+ }
+ }
+ }
+}
+
fn fitEnd(text: []const u8, start: usize, width: usize) usize {
var end = start;
var used: usize = 0;