summaryrefslogtreecommitdiff
path: root/src/modal.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-26 13:27:46 -0300
committerGabriel Schneider <[email protected]>2026-08-27 09:47:39 -0300
commit11f380f6d7222f2cad93c2cdf13701ea1f903d47 (patch)
tree803194ee5853a6b4cda93f90a95e28d1f02e69ae /src/modal.zig
parentfbc194068687e49a8490c85c9f1257a2f2bb9079 (diff)
downloadpardes-11f380f6d7222f2cad93c2cdf13701ea1f903d47.tar.gz
pardes-11f380f6d7222f2cad93c2cdf13701ea1f903d47.zip
One core behind N frontends, the board's own runner moved in, and every board cap on one screen
## The wire is the effect stream, not a new protocol `pardes --detach` leaves a core running with no terminal; `pardes --attach` is a frontend that owns a terminal and a socket and nothing else. N frontends on one core all look at the same screen — `screen -x`, not N sessions. The codec (`src/detached/wire.zig`) carries exactly one `Event` or one `Host.VTable` call per message. That is not a coincidence and it is why there is no third vocabulary to keep in step: the core's IO seam was already a struct of function pointers with plain-data arguments, so a socket is a legal implementation of it. `nested.zig`'s socket could not be reused — it carries a builtin command line, and a command line cannot carry a frame. ARCHITECTURE-NEUTRAL on purpose, not as decoration. The frontend on the far end may be riscv32-freestanding on the ESP32-P4 while the core is x86_64 Linux, so every field is an explicit little-endian fixed width and no message is a blit of a native struct. A protocol that only works between two builds of the same compiler would have thrown away the one frontend that motivated it. ## The board comes in; its toolchain stays out `src/p4.zig` becomes `src/esp32p4.zig`, and the pardes half of `../05-zig-p4` — the vaxis-over- serial runner, the UART editor terminal, the keystroke rescue ring, the on-die test suite — moves into `src/esp32p4/`. `build.zig.zon` gains `.zig_p4 = .{ .path = "../05-zig-p4" }`, so `zig build -Dplatform=esp32p4 -Desp32p4-firmware` builds, flashes, monitors and self-tests the board from this repo's `build.zig`. The DIVISION is the point. What moved is what only pardes wants: the runner that drives a pardes core over a serial line. What stayed is everything a second project would also want — the HAL, the register/radio/oracle layers, the linker script, `_start`. `zig_p4` declares no dependencies of its own and its `build()` early-returns when it is not the root package, so this costs the package graph exactly zero packages and the editor's own builds nothing at all. ## limits.zig: nine forgettable places become one budget Nine `platform == .esp32p4` capacity tests lived in nine files. They were never nine decisions — they are ONE decision, how much memory this build may spend, taken nine times where no reader could see the total. `src/limits.zig` puts the whole budget on one screen with every cap named against what it is measured against, derived from two booleans. The payoff is testability on a machine that is not the board: the caps are ordinary comptime values, so a host build can be compiled against the board's numbers and the parking, eviction and clamping paths a 240 KiB core takes get exercised by the normal test suite instead of only over a UART. ## A bare `zig build` `zig build` with no arguments now builds the tty and GUI binaries and installs them into `~/.local/bin`, and says so once on stdout with the flag that overrides it. The old default built one binary into `zig-out` — a path nothing on a `PATH` ever looks at, which made "build it" and "use it" two different commands for no reason.
Diffstat (limited to 'src/modal.zig')
-rw-r--r--src/modal.zig121
1 files changed, 119 insertions, 2 deletions
diff --git a/src/modal.zig b/src/modal.zig
index 953aa900..11eae743 100644
--- a/src/modal.zig
+++ b/src/modal.zig
@@ -886,7 +886,15 @@ pub fn nextGrapheme(text: []const u8, off: usize) usize {
if (off >= text.len) return text.len;
// The editor's own offsets are already boundaries. Keep the overwhelmingly
// common ASCII path O(1); only repair a continuation-byte input here.
- if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1;
+ //
+ // GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a
+ // following LF into the same cluster. `graphemeStart` spells that exclusion
+ // out (:875) and this did not, so the two disagreed about a CRLF file by
+ // exactly one byte — a head stepped onto the offset between CR and LF and
+ // `graphemeStart` then repaired it back onto the CR. Excluded here for the
+ // same reason and in the same words; everything else ASCII is still O(1).
+ if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80) and
+ !(text[off] == '\r' and off + 1 < text.len and text[off + 1] == '\n')) return off + 1;
var start = off;
while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1;
if (start != off) start = graphemeStart(text, off);
@@ -903,7 +911,10 @@ pub fn prevGrapheme(text: []const u8, off: usize) usize {
bounded = repaired;
}
if (bounded == 0) return 0;
- if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1;
+ // ...and the same GB3 exclusion, from the other side: a CR before this LF
+ // means the cluster starts one byte earlier than the fast path would say.
+ if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80) and
+ !(bounded >= 2 and text[bounded - 2] == '\r' and text[bounded - 1] == '\n')) return bounded - 1;
// Graphemes cannot cross a line break. Restrict the forward segmentation
// needed for a reverse step to the current line instead of rescanning the
// complete buffer.
@@ -1470,6 +1481,112 @@ test "extended grapheme boundaries cover combining emoji flag and CJK text" {
try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3));
}
+test "the ASCII arms of graphemeStart and nextGrapheme agree with the UAX #29 walk" {
+ // Both functions answer ASCII from arithmetic and hand everything else to the segmenter. The
+ // guard is a claim about UAX #29 (an ASCII scalar is its own cluster unless the next scalar
+ // extends it, and every extender is non-ASCII), so pin it against the walk it skips rather
+ // than against transcribed offsets: same text, both routes, every offset including past the end.
+ const H = struct {
+ // `graphemeStart` with the ASCII arm deleted — nothing else changed.
+ fn start(text: []const u8, off: usize) usize {
+ const bounded = @min(off, text.len);
+ if (bounded == text.len) return text.len;
+ var it = uucode.grapheme.utf8Iterator(text);
+ while (it.nextGrapheme()) |g| {
+ if (bounded < g.end) return g.start;
+ }
+ return text.len;
+ }
+ // `nextGrapheme` with the ASCII arm deleted.
+ fn next(text: []const u8, off: usize) usize {
+ if (off >= text.len) return text.len;
+ var s = off;
+ while (s > 0 and (text[s] & 0xC0) == 0x80) s -= 1;
+ if (s != off) s = start(text, off);
+ var it = uucode.grapheme.utf8Iterator(text[s..]);
+ const g = it.nextGrapheme() orelse return @min(s + 1, text.len);
+ return s + g.end;
+ }
+ fn check(text: []const u8) !void {
+ var off: usize = 0;
+ while (off <= text.len + 2) : (off += 1) {
+ std.testing.expectEqual(start(text, off), graphemeStart(text, off)) catch |e| {
+ std.debug.print("graphemeStart({any}, {d})\n", .{ text, off });
+ return e;
+ };
+ std.testing.expectEqual(next(text, off), nextGrapheme(text, off)) catch |e| {
+ std.debug.print("nextGrapheme({any}, {d})\n", .{ text, off });
+ return e;
+ };
+ }
+ }
+ };
+
+ // Scalars that extend a preceding ASCII base into ONE cluster, which is the whole reason the
+ // fast path inspects its neighbour: a combining mark, a ZWJ sequence, a spacing mark
+ // (Devanagari visarga), a variation selector. Plus wide glyphs, a regional-indicator pair,
+ // and three shapes of invalid UTF-8 the segmenter must still be trusted with: a bad start
+ // byte, a truncated tail, a bad continuation.
+ const neighbours = [_][]const u8{
+ "", "a", "\u{301}", "\u{200d}\u{1f680}",
+ "\u{903}", "\u{fe0f}", "\u{20e3}", "\u{4e16}\u{754c}",
+ "\u{1f642}", "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8",
+ "\xe4\x28\xb8",
+ };
+ // Every byte the range test can see, ASCII and not: 0x20..0x7e take the fast path, and \t, \r,
+ // the rest of the C0 controls and DEL are excluded by it and must still reach the same answer.
+ var buf: [16]u8 = undefined;
+ var b: u8 = 0;
+ while (b < 0x80) : (b += 1) {
+ buf[0] = b;
+ for (neighbours) |tail| {
+ @memcpy(buf[1..][0..tail.len], tail);
+ try H.check(buf[0 .. 1 + tail.len]);
+ // ...and the same byte as a follower, so a boundary is probed from both sides.
+ @memcpy(buf[0..tail.len], tail);
+ buf[tail.len] = b;
+ try H.check(buf[0 .. tail.len + 1]);
+ }
+ }
+
+ // Text that has no CR-LF pair in it: GB3 is the one ASCII-only rule that joins two clusters,
+ // and it gets its own test below because it is the single exclusion every fast path has to
+ // carry by hand.
+ for ([_][]const u8{ "a\r", "\ra", "\n\r", "a\rb\nc" }) |text| try H.check(text);
+
+ // Mixed text long enough that a fast-path run starts, ends and restarts inside one string.
+ try H.check("plain ascii then \u{4e16}\u{754c} then e\u{301} then more ascii");
+}
+
+// GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a following LF into the same
+// cluster. Each of the three steppers carries that exclusion separately - `graphemeStart` at :875,
+// `nextGrapheme`'s ASCII arm at :896, `prevGrapheme`'s at :916 - so nothing but a test keeps them
+// agreeing. The invariant is that all three answer the same CRLF boundary: for every cluster the
+// segmenter reports, `graphemeStart` maps its start to itself, `nextGrapheme` maps that start to
+// its end, and `prevGrapheme` maps its end back to the start.
+//
+// This was a live bug: `nextGrapheme` and `prevGrapheme` stepped exactly one byte whenever the
+// byte at the offset and its neighbour were ASCII, so on a CRLF file the flat-buffer range engine
+// could step a head to offset 1 and `graphemeStart` would repair that same offset back to 0. Both
+// arms now spell the exclusion out, and this test is what holds them there.
+test "GB3 keeps CR-LF one cluster for every grapheme step" {
+ const text = "a\r\nb";
+ // The reference: the same segmentation the slow arms of these functions run.
+ var it = uucode.grapheme.utf8Iterator(text);
+ var starts: [8]usize = undefined;
+ var ends: [8]usize = undefined;
+ var n: usize = 0;
+ while (it.nextGrapheme()) |g| : (n += 1) {
+ starts[n] = g.start;
+ ends[n] = g.end;
+ }
+ for (starts[0..n], ends[0..n]) |start, end| {
+ try std.testing.expectEqual(start, graphemeStart(text, start));
+ try std.testing.expectEqual(end, nextGrapheme(text, start));
+ try std.testing.expectEqual(start, prevGrapheme(text, end));
+ }
+}
+
test "Unicode find and word motion stay on grapheme boundaries" {
const lines = [_][]const u8{"\u{e9}x\u{e9}"};
try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'é', true, false, 1).?);