diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-25 22:10:36 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-25 22:39:09 -0300 |
| commit | 97f9ba329081c23d3ac9c9a46338e0178e692742 (patch) | |
| tree | cde6dda534ddd8e16eb2143eea497ae22b0f28fb | |
| parent | 209b48a36db527904ddc4b436f79df45cea3ab3e (diff) | |
| download | pardes-97f9ba329081c23d3ac9c9a46338e0178e692742.tar.gz pardes-97f9ba329081c23d3ac9c9a46338e0178e692742.zip | |
Consume an ASCII run in fitEnd instead of asking twice per character
`fitEnd` decides where a wrapped row breaks, and it asked two function calls per
character to learn what arithmetic knows. `modal.nextGrapheme` and
`graphemeDisplayWidth` each already answer ASCII in constant time - that was earlier
work - but they answer once per character, and a 640-column line asks 640 times.
A printable ASCII byte whose successor is also ASCII is a complete grapheme cluster one
column wide. That is the same guard, for the same reason, as the three fast paths already
in `Surface.print`, `modal.nextGrapheme` and `graphemeDisplayWidth`: every rule that
could join an ASCII base into a longer cluster - Extend, ZWJ, SpacingMark, Prepend,
Regional_Indicator - is spelled with non-ASCII scalars. Tabs and the C0 controls are
excluded by the range test and keep the general path, as does anything wide.
Measured on the die: 21 us of a 640-character keystroke, 5 us at 160. Small, and reported
as small - the interesting part is that it is small, because it says the per-character
grapheme walk was NOT where a long line's cost lives.
The test pins the fast path to the general walk it replaces rather than to transcribed
expectations: same inputs through both routes, every start offset, every width from zero
to past the end, over strings chosen to land the boundary inside a combining sequence, a
wide glyph, a regional-indicator pair, a tab and a CR. A break that moved by one column
would move text on screen, so this is the invariant worth holding.
| -rw-r--r-- | src/file_pane.zig | 62 |
1 files changed, 62 insertions, 0 deletions
diff --git a/src/file_pane.zig b/src/file_pane.zig index 1c6d5e2f..e0150de2 100644 --- a/src/file_pane.zig +++ b/src/file_pane.zig @@ -223,6 +223,19 @@ test "display columns map complete Unicode graphemes" { fn fitEnd(text: []const u8, start: usize, width: usize) usize { var end = start; var used: usize = 0; + // ASCII RUN. This is the loop a wrapped line pays per character, and it asks two function calls + // to learn what arithmetic knows: `modal.nextGrapheme` and `graphemeDisplayWidth` each answer + // ASCII in constant time, but they answer once per character and a 640-column line asks 640 + // times. A printable ASCII byte whose successor is also ASCII is a complete grapheme cluster one + // column wide - the same guard, and the same reason, as `Surface.print` and `modal.nextGrapheme` + // - so consume the run here and leave anything else to the general path below. + while (used < width and end < text.len) { + const b = text[end]; + if (b < 0x20 or b >= 0x7f) break; + if (end + 1 < text.len and text[end + 1] >= 0x80) break; + used += 1; + end += 1; + } while (end < text.len) { const next_end = modal.nextGrapheme(text, end); const next_used = used +| graphemeDisplayWidth(text[end..next_end]); @@ -233,6 +246,55 @@ fn fitEnd(text: []const u8, start: usize, width: usize) usize { return end; } +test "the ASCII run in fitEnd cuts where the grapheme walk would" { + // fitEnd decides where a wrapped row BREAKS, so a fast path that is off by one column moves + // text on screen. This pins it to the general walk it replaces rather than to a transcribed + // expectation: same inputs, both routes, every width from 0 past the end of the string. + const reference = struct { + fn fitEnd(text: []const u8, start: usize, width: usize) usize { + var end = start; + var used: usize = 0; + while (end < text.len) { + const next_end = modal.nextGrapheme(text, end); + const next_used = used +| graphemeDisplayWidth(text[end..next_end]); + if (next_used > width) return if (end == start) next_end else end; + used = next_used; + end = next_end; + } + return end; + } + }.fitEnd; + + const cases = [_][]const u8{ + "", + "hello world", + // the fast path must hand over at the first non-ASCII byte, mid-run + "abc\u{00e9}def", + // a wide glyph is two columns, so a width boundary can land inside it + "ab\u{4e16}\u{754c}cd", + // a cluster the fast path must not split + "a\u{0301}bc", + // tabs and controls are excluded from the fast path by the range test + "ab\tcd", + "ab\rcd", + // an ASCII byte followed by a continuation byte is NOT its own cluster + "e\u{0301}x", + "\u{1f1e6}\u{1f1e7}ok", + }; + for (cases) |text| { + var width: usize = 0; + while (width <= text.len + 3) : (width += 1) { + var start: usize = 0; + while (start <= text.len) : (start += 1) { + try std.testing.expectEqual( + reference(text, start, width), + fitEnd(text, start, width), + ); + } + } + } +} + pub fn lineCount(content: []const u8) usize { return std.mem.count(u8, content, "\n") + 1; } |
