diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-29 08:11:20 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-10-01 00:12:16 -0300 |
| commit | 6c10e2b37ee7b95485fb86941de32991a1732892 (patch) | |
| tree | a26e3bc3f94318b242cddd75c58f715e749f197e /src/ninep/events.zig | |
| parent | f7e3441698a4621bec0d7c14eb45303753ffbead (diff) | |
| download | pardes-6c10e2b37ee7b95485fb86941de32991a1732892.tar.gz pardes-6c10e2b37ee7b95485fb86941de32991a1732892.zip | |
A name write is one name, refused with its reason otherwise; a log record is one line of UTF-8
A name took a second line, DEL, a C1 control or bytes that are not UTF-8,
which then went into /index and /log as they were and split or garbled a
reader's lines. `name` strips one trailing newline and refuses the rest,
saying which (`bad character in file name: not UTF-8`); and the log turns
DEL and C1 into spaces, as it did control characters, and escapes bytes
that are not UTF-8 as `\xNN`.
Co-Authored-By: Claude Opus 5.5 <[email protected]>
Diffstat (limited to 'src/ninep/events.zig')
| -rw-r--r-- | src/ninep/events.zig | 55 |
1 files changed, 50 insertions, 5 deletions
diff --git a/src/ninep/events.zig b/src/ninep/events.zig index 98799774..d7079c5e 100644 --- a/src/ninep/events.zig +++ b/src/ninep/events.zig @@ -292,11 +292,13 @@ pub fn dropMessage(p: *Pardes, text: []const u8) void { /// counts that one instead: `<record> (x2)`, the count being every time it /// was said. One a follower has already read is not rewritten: the repeat /// is a new line carrying the running count, `(x3)`. -fn pushCounting(p: *Pardes, record: []u8) void { +fn pushCounting(p: *Pardes, raw: []u8) void { // A client retrying a write that fails the same way would fill the ring // with one line, so a repeat of the newest record is that record counted, // `(x3)`, as +Messages counts its repeats; unless a follower has read it // already and so waits on the repeat as a line of its own. + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); const last = newest(p) orelse return pushLog(p, record); var said = last.text; var times: u32 = 1; @@ -340,10 +342,47 @@ fn followerRead(p: *Pardes, seq: u64) bool { /// The log is one ring that records whether or not anyone reads it. A record /// is one line: a newline in a message or a name would read as two records. -fn pushLog(p: *Pardes, record: []u8) void { - for (record[0 .. record.len - 1]) |*c| if (c.* < ' ') { - c.* = ' '; - }; +/// A record as one line of text: a control character, DEL or a C1 control +/// (U+0080-U+009F) is a space each, and a byte that is not UTF-8 is +/// written `\xNN`, so a reader splitting on newlines and decoding UTF-8 +/// never trips. The record's own newline, last, is kept. +fn sanitize(record: []const u8, out: []u8) []u8 { + var w: usize = 0; + var i: usize = 0; + const body = record[0 .. record.len - 1]; + while (i < body.len and w + 4 < out.len) { + const c = body[i]; + if (c < ' ' or c == 0x7f) { + out[w] = ' '; + w += 1; + i += 1; + continue; + } + const n = std.unicode.utf8ByteSequenceLength(c) catch 0; + if (n == 0 or i + n > body.len or !std.unicode.utf8ValidateSlice(body[i .. i + n])) { + _ = std.fmt.bufPrint(out[w..], "\\x{x:0>2}", .{c}) catch break; + w += 4; + i += 1; + continue; + } + if (n == 2 and c == 0xC2 and body[i + 1] <= 0x9F) { + out[w] = ' '; + w += 1; + i += 2; + continue; + } + if (w + n + 1 > out.len) break; + @memcpy(out[w..][0..n], body[i..][0..n]); + w += n; + i += n; + } + out[w] = '\n'; + return out[0 .. w + 1]; +} + +fn pushLog(p: *Pardes, raw: []u8) void { + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); // One record larger than the ring would push every other out and then // not fit itself; cut it to what fits instead, on a character boundary. var end = @min(record.len, p.fs.log.cap - 4) - 1; @@ -1244,6 +1283,12 @@ test "a long msg record is cut between words at its cap, with an ellipsis" { _ = call(p, .{ .tag = 3, .op = .release, .node = log, .handle = h }); } +test "a record is one line of UTF-8: DEL and C1 are spaces, bytes not UTF-8 are escaped" { + var out: [64]u8 = undefined; + try testing.expectEqualStrings("msg - a b c \\xff d\n", sanitize("msg - a\x7fb\xc2\x85c \xff d\n", &out)); + try testing.expectEqualStrings("msg - caf\xc3\xa9\n", sanitize("msg - caf\xc3\xa9\n", &out)); +} + test "a long err record is cut between words, with an ellipsis" { const p = try withFile(testing.allocator, "x\n"); defer p.deinit(); |
