summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-25 22:10:36 -0300
committerGabriel Schneider <[email protected]>2026-08-25 22:39:09 -0300
commit97f9ba329081c23d3ac9c9a46338e0178e692742 (patch)
treecde6dda534ddd8e16eb2143eea497ae22b0f28fb
parent209b48a36db527904ddc4b436f79df45cea3ab3e (diff)
downloadpardes-97f9ba329081c23d3ac9c9a46338e0178e692742.tar.gz
pardes-97f9ba329081c23d3ac9c9a46338e0178e692742.zip
Consume an ASCII run in fitEnd instead of asking twice per character
`fitEnd` decides where a wrapped row breaks, and it asked two function calls per character to learn what arithmetic knows. `modal.nextGrapheme` and `graphemeDisplayWidth` each already answer ASCII in constant time - that was earlier work - but they answer once per character, and a 640-column line asks 640 times. A printable ASCII byte whose successor is also ASCII is a complete grapheme cluster one column wide. That is the same guard, for the same reason, as the three fast paths already in `Surface.print`, `modal.nextGrapheme` and `graphemeDisplayWidth`: every rule that could join an ASCII base into a longer cluster - Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator - is spelled with non-ASCII scalars. Tabs and the C0 controls are excluded by the range test and keep the general path, as does anything wide. Measured on the die: 21 us of a 640-character keystroke, 5 us at 160. Small, and reported as small - the interesting part is that it is small, because it says the per-character grapheme walk was NOT where a long line's cost lives. The test pins the fast path to the general walk it replaces rather than to transcribed expectations: same inputs through both routes, every start offset, every width from zero to past the end, over strings chosen to land the boundary inside a combining sequence, a wide glyph, a regional-indicator pair, a tab and a CR. A break that moved by one column would move text on screen, so this is the invariant worth holding.
-rw-r--r--src/file_pane.zig62
1 files changed, 62 insertions, 0 deletions
diff --git a/src/file_pane.zig b/src/file_pane.zig
index 1c6d5e2f..e0150de2 100644
--- a/src/file_pane.zig
+++ b/src/file_pane.zig
@@ -223,6 +223,19 @@ test "display columns map complete Unicode graphemes" {
fn fitEnd(text: []const u8, start: usize, width: usize) usize {
var end = start;
var used: usize = 0;
+ // ASCII RUN. This is the loop a wrapped line pays per character, and it asks two function calls
+ // to learn what arithmetic knows: `modal.nextGrapheme` and `graphemeDisplayWidth` each answer
+ // ASCII in constant time, but they answer once per character and a 640-column line asks 640
+ // times. A printable ASCII byte whose successor is also ASCII is a complete grapheme cluster one
+ // column wide - the same guard, and the same reason, as `Surface.print` and `modal.nextGrapheme`
+ // - so consume the run here and leave anything else to the general path below.
+ while (used < width and end < text.len) {
+ const b = text[end];
+ if (b < 0x20 or b >= 0x7f) break;
+ if (end + 1 < text.len and text[end + 1] >= 0x80) break;
+ used += 1;
+ end += 1;
+ }
while (end < text.len) {
const next_end = modal.nextGrapheme(text, end);
const next_used = used +| graphemeDisplayWidth(text[end..next_end]);
@@ -233,6 +246,55 @@ fn fitEnd(text: []const u8, start: usize, width: usize) usize {
return end;
}
+test "the ASCII run in fitEnd cuts where the grapheme walk would" {
+ // fitEnd decides where a wrapped row BREAKS, so a fast path that is off by one column moves
+ // text on screen. This pins it to the general walk it replaces rather than to a transcribed
+ // expectation: same inputs, both routes, every width from 0 past the end of the string.
+ const reference = struct {
+ fn fitEnd(text: []const u8, start: usize, width: usize) usize {
+ var end = start;
+ var used: usize = 0;
+ while (end < text.len) {
+ const next_end = modal.nextGrapheme(text, end);
+ const next_used = used +| graphemeDisplayWidth(text[end..next_end]);
+ if (next_used > width) return if (end == start) next_end else end;
+ used = next_used;
+ end = next_end;
+ }
+ return end;
+ }
+ }.fitEnd;
+
+ const cases = [_][]const u8{
+ "",
+ "hello world",
+ // the fast path must hand over at the first non-ASCII byte, mid-run
+ "abc\u{00e9}def",
+ // a wide glyph is two columns, so a width boundary can land inside it
+ "ab\u{4e16}\u{754c}cd",
+ // a cluster the fast path must not split
+ "a\u{0301}bc",
+ // tabs and controls are excluded from the fast path by the range test
+ "ab\tcd",
+ "ab\rcd",
+ // an ASCII byte followed by a continuation byte is NOT its own cluster
+ "e\u{0301}x",
+ "\u{1f1e6}\u{1f1e7}ok",
+ };
+ for (cases) |text| {
+ var width: usize = 0;
+ while (width <= text.len + 3) : (width += 1) {
+ var start: usize = 0;
+ while (start <= text.len) : (start += 1) {
+ try std.testing.expectEqual(
+ reference(text, start, width),
+ fitEnd(text, start, width),
+ );
+ }
+ }
+ }
+}
+
pub fn lineCount(content: []const u8) usize {
return std.mem.count(u8, content, "\n") + 1;
}