From 11f380f6d7222f2cad93c2cdf13701ea1f903d47 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Wed, 26 Aug 2026 13:27:46 -0300 Subject: One core behind N frontends, the board's own runner moved in, and every board cap on one screen MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## The wire is the effect stream, not a new protocol `pardes --detach` leaves a core running with no terminal; `pardes --attach` is a frontend that owns a terminal and a socket and nothing else. N frontends on one core all look at the same screen — `screen -x`, not N sessions. The codec (`src/detached/wire.zig`) carries exactly one `Event` or one `Host.VTable` call per message. That is not a coincidence and it is why there is no third vocabulary to keep in step: the core's IO seam was already a struct of function pointers with plain-data arguments, so a socket is a legal implementation of it. `nested.zig`'s socket could not be reused — it carries a builtin command line, and a command line cannot carry a frame. ARCHITECTURE-NEUTRAL on purpose, not as decoration. The frontend on the far end may be riscv32-freestanding on the ESP32-P4 while the core is x86_64 Linux, so every field is an explicit little-endian fixed width and no message is a blit of a native struct. A protocol that only works between two builds of the same compiler would have thrown away the one frontend that motivated it. ## The board comes in; its toolchain stays out `src/p4.zig` becomes `src/esp32p4.zig`, and the pardes half of `../05-zig-p4` — the vaxis-over- serial runner, the UART editor terminal, the keystroke rescue ring, the on-die test suite — moves into `src/esp32p4/`. `build.zig.zon` gains `.zig_p4 = .{ .path = "../05-zig-p4" }`, so `zig build -Dplatform=esp32p4 -Desp32p4-firmware` builds, flashes, monitors and self-tests the board from this repo's `build.zig`. The DIVISION is the point. What moved is what only pardes wants: the runner that drives a pardes core over a serial line. What stayed is everything a second project would also want — the HAL, the register/radio/oracle layers, the linker script, `_start`. `zig_p4` declares no dependencies of its own and its `build()` early-returns when it is not the root package, so this costs the package graph exactly zero packages and the editor's own builds nothing at all. ## limits.zig: nine forgettable places become one budget Nine `platform == .esp32p4` capacity tests lived in nine files. They were never nine decisions — they are ONE decision, how much memory this build may spend, taken nine times where no reader could see the total. `src/limits.zig` puts the whole budget on one screen with every cap named against what it is measured against, derived from two booleans. The payoff is testability on a machine that is not the board: the caps are ordinary comptime values, so a host build can be compiled against the board's numbers and the parking, eviction and clamping paths a 240 KiB core takes get exercised by the normal test suite instead of only over a UART. ## A bare `zig build` `zig build` with no arguments now builds the tty and GUI binaries and installs them into `~/.local/bin`, and says so once on stdout with the flag that overrides it. The old default built one binary into `zig-out` — a path nothing on a `PATH` ever looks at, which made "build it" and "use it" two different commands for no reason. --- src/modal.zig | 121 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 119 insertions(+), 2 deletions(-) (limited to 'src/modal.zig') diff --git a/src/modal.zig b/src/modal.zig index 953aa900..11eae743 100644 --- a/src/modal.zig +++ b/src/modal.zig @@ -886,7 +886,15 @@ pub fn nextGrapheme(text: []const u8, off: usize) usize { if (off >= text.len) return text.len; // The editor's own offsets are already boundaries. Keep the overwhelmingly // common ASCII path O(1); only repair a continuation-byte input here. - if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80)) return off + 1; + // + // GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a + // following LF into the same cluster. `graphemeStart` spells that exclusion + // out (:875) and this did not, so the two disagreed about a CRLF file by + // exactly one byte — a head stepped onto the offset between CR and LF and + // `graphemeStart` then repaired it back onto the CR. Excluded here for the + // same reason and in the same words; everything else ASCII is still O(1). + if (text[off] < 0x80 and (off + 1 == text.len or text[off + 1] < 0x80) and + !(text[off] == '\r' and off + 1 < text.len and text[off + 1] == '\n')) return off + 1; var start = off; while (start > 0 and (text[start] & 0xC0) == 0x80) start -= 1; if (start != off) start = graphemeStart(text, off); @@ -903,7 +911,10 @@ pub fn prevGrapheme(text: []const u8, off: usize) usize { bounded = repaired; } if (bounded == 0) return 0; - if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80)) return bounded - 1; + // ...and the same GB3 exclusion, from the other side: a CR before this LF + // means the cluster starts one byte earlier than the fast path would say. + if (text[bounded - 1] < 0x80 and (bounded == 1 or text[bounded - 2] < 0x80) and + !(bounded >= 2 and text[bounded - 2] == '\r' and text[bounded - 1] == '\n')) return bounded - 1; // Graphemes cannot cross a line break. Restrict the forward segmentation // needed for a reverse step to the current line instead of rescanning the // complete buffer. @@ -1470,6 +1481,112 @@ test "extended grapheme boundaries cover combining emoji flag and CJK text" { try std.testing.expectEqual(@as(usize, 19), graphemeAtColumn(text, 3)); } +test "the ASCII arms of graphemeStart and nextGrapheme agree with the UAX #29 walk" { + // Both functions answer ASCII from arithmetic and hand everything else to the segmenter. The + // guard is a claim about UAX #29 (an ASCII scalar is its own cluster unless the next scalar + // extends it, and every extender is non-ASCII), so pin it against the walk it skips rather + // than against transcribed offsets: same text, both routes, every offset including past the end. + const H = struct { + // `graphemeStart` with the ASCII arm deleted — nothing else changed. + fn start(text: []const u8, off: usize) usize { + const bounded = @min(off, text.len); + if (bounded == text.len) return text.len; + var it = uucode.grapheme.utf8Iterator(text); + while (it.nextGrapheme()) |g| { + if (bounded < g.end) return g.start; + } + return text.len; + } + // `nextGrapheme` with the ASCII arm deleted. + fn next(text: []const u8, off: usize) usize { + if (off >= text.len) return text.len; + var s = off; + while (s > 0 and (text[s] & 0xC0) == 0x80) s -= 1; + if (s != off) s = start(text, off); + var it = uucode.grapheme.utf8Iterator(text[s..]); + const g = it.nextGrapheme() orelse return @min(s + 1, text.len); + return s + g.end; + } + fn check(text: []const u8) !void { + var off: usize = 0; + while (off <= text.len + 2) : (off += 1) { + std.testing.expectEqual(start(text, off), graphemeStart(text, off)) catch |e| { + std.debug.print("graphemeStart({any}, {d})\n", .{ text, off }); + return e; + }; + std.testing.expectEqual(next(text, off), nextGrapheme(text, off)) catch |e| { + std.debug.print("nextGrapheme({any}, {d})\n", .{ text, off }); + return e; + }; + } + } + }; + + // Scalars that extend a preceding ASCII base into ONE cluster, which is the whole reason the + // fast path inspects its neighbour: a combining mark, a ZWJ sequence, a spacing mark + // (Devanagari visarga), a variation selector. Plus wide glyphs, a regional-indicator pair, + // and three shapes of invalid UTF-8 the segmenter must still be trusted with: a bad start + // byte, a truncated tail, a bad continuation. + const neighbours = [_][]const u8{ + "", "a", "\u{301}", "\u{200d}\u{1f680}", + "\u{903}", "\u{fe0f}", "\u{20e3}", "\u{4e16}\u{754c}", + "\u{1f642}", "\u{1f1e6}\u{1f1e7}", "\xff", "\xe4\xb8", + "\xe4\x28\xb8", + }; + // Every byte the range test can see, ASCII and not: 0x20..0x7e take the fast path, and \t, \r, + // the rest of the C0 controls and DEL are excluded by it and must still reach the same answer. + var buf: [16]u8 = undefined; + var b: u8 = 0; + while (b < 0x80) : (b += 1) { + buf[0] = b; + for (neighbours) |tail| { + @memcpy(buf[1..][0..tail.len], tail); + try H.check(buf[0 .. 1 + tail.len]); + // ...and the same byte as a follower, so a boundary is probed from both sides. + @memcpy(buf[0..tail.len], tail); + buf[tail.len] = b; + try H.check(buf[0 .. tail.len + 1]); + } + } + + // Text that has no CR-LF pair in it: GB3 is the one ASCII-only rule that joins two clusters, + // and it gets its own test below because it is the single exclusion every fast path has to + // carry by hand. + for ([_][]const u8{ "a\r", "\ra", "\n\r", "a\rb\nc" }) |text| try H.check(text); + + // Mixed text long enough that a fast-path run starts, ends and restarts inside one string. + try H.check("plain ascii then \u{4e16}\u{754c} then e\u{301} then more ascii"); +} + +// GB3 is the one UAX #29 rule that joins two ASCII scalars: CR takes a following LF into the same +// cluster. Each of the three steppers carries that exclusion separately - `graphemeStart` at :875, +// `nextGrapheme`'s ASCII arm at :896, `prevGrapheme`'s at :916 - so nothing but a test keeps them +// agreeing. The invariant is that all three answer the same CRLF boundary: for every cluster the +// segmenter reports, `graphemeStart` maps its start to itself, `nextGrapheme` maps that start to +// its end, and `prevGrapheme` maps its end back to the start. +// +// This was a live bug: `nextGrapheme` and `prevGrapheme` stepped exactly one byte whenever the +// byte at the offset and its neighbour were ASCII, so on a CRLF file the flat-buffer range engine +// could step a head to offset 1 and `graphemeStart` would repair that same offset back to 0. Both +// arms now spell the exclusion out, and this test is what holds them there. +test "GB3 keeps CR-LF one cluster for every grapheme step" { + const text = "a\r\nb"; + // The reference: the same segmentation the slow arms of these functions run. + var it = uucode.grapheme.utf8Iterator(text); + var starts: [8]usize = undefined; + var ends: [8]usize = undefined; + var n: usize = 0; + while (it.nextGrapheme()) |g| : (n += 1) { + starts[n] = g.start; + ends[n] = g.end; + } + for (starts[0..n], ends[0..n]) |start, end| { + try std.testing.expectEqual(start, graphemeStart(text, start)); + try std.testing.expectEqual(end, nextGrapheme(text, start)); + try std.testing.expectEqual(start, prevGrapheme(text, end)); + } +} + test "Unicode find and word motion stay on grapheme boundaries" { const lines = [_][]const u8{"\u{e9}x\u{e9}"}; try std.testing.expectEqual(Cursor{ .row = 0, .col = 3 }, findChar(&lines, .{ .row = 0, .col = 0 }, 'é', true, false, 1).?); -- cgit v1.3