diff options
Diffstat (limited to 'src/ninep/events.zig')
| -rw-r--r-- | src/ninep/events.zig | 55 |
1 files changed, 50 insertions, 5 deletions
diff --git a/src/ninep/events.zig b/src/ninep/events.zig index 98799774..d7079c5e 100644 --- a/src/ninep/events.zig +++ b/src/ninep/events.zig @@ -292,11 +292,13 @@ pub fn dropMessage(p: *Pardes, text: []const u8) void { /// counts that one instead: `<record> (x2)`, the count being every time it /// was said. One a follower has already read is not rewritten: the repeat /// is a new line carrying the running count, `(x3)`. -fn pushCounting(p: *Pardes, record: []u8) void { +fn pushCounting(p: *Pardes, raw: []u8) void { // A client retrying a write that fails the same way would fill the ring // with one line, so a repeat of the newest record is that record counted, // `(x3)`, as +Messages counts its repeats; unless a follower has read it // already and so waits on the repeat as a line of its own. + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); const last = newest(p) orelse return pushLog(p, record); var said = last.text; var times: u32 = 1; @@ -340,10 +342,47 @@ fn followerRead(p: *Pardes, seq: u64) bool { /// The log is one ring that records whether or not anyone reads it. A record /// is one line: a newline in a message or a name would read as two records. -fn pushLog(p: *Pardes, record: []u8) void { - for (record[0 .. record.len - 1]) |*c| if (c.* < ' ') { - c.* = ' '; - }; +/// A record as one line of text: a control character, DEL or a C1 control +/// (U+0080-U+009F) is a space each, and a byte that is not UTF-8 is +/// written `\xNN`, so a reader splitting on newlines and decoding UTF-8 +/// never trips. The record's own newline, last, is kept. +fn sanitize(record: []const u8, out: []u8) []u8 { + var w: usize = 0; + var i: usize = 0; + const body = record[0 .. record.len - 1]; + while (i < body.len and w + 4 < out.len) { + const c = body[i]; + if (c < ' ' or c == 0x7f) { + out[w] = ' '; + w += 1; + i += 1; + continue; + } + const n = std.unicode.utf8ByteSequenceLength(c) catch 0; + if (n == 0 or i + n > body.len or !std.unicode.utf8ValidateSlice(body[i .. i + n])) { + _ = std.fmt.bufPrint(out[w..], "\\x{x:0>2}", .{c}) catch break; + w += 4; + i += 1; + continue; + } + if (n == 2 and c == 0xC2 and body[i + 1] <= 0x9F) { + out[w] = ' '; + w += 1; + i += 2; + continue; + } + if (w + n + 1 > out.len) break; + @memcpy(out[w..][0..n], body[i..][0..n]); + w += n; + i += n; + } + out[w] = '\n'; + return out[0 .. w + 1]; +} + +fn pushLog(p: *Pardes, raw: []u8) void { + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); // One record larger than the ring would push every other out and then // not fit itself; cut it to what fits instead, on a character boundary. var end = @min(record.len, p.fs.log.cap - 4) - 1; @@ -1244,6 +1283,12 @@ test "a long msg record is cut between words at its cap, with an ellipsis" { _ = call(p, .{ .tag = 3, .op = .release, .node = log, .handle = h }); } +test "a record is one line of UTF-8: DEL and C1 are spaces, bytes not UTF-8 are escaped" { + var out: [64]u8 = undefined; + try testing.expectEqualStrings("msg - a b c \\xff d\n", sanitize("msg - a\x7fb\xc2\x85c \xff d\n", &out)); + try testing.expectEqualStrings("msg - caf\xc3\xa9\n", sanitize("msg - caf\xc3\xa9\n", &out)); +} + test "a long err record is cut between words, with an ellipsis" { const p = try withFile(testing.allocator, "x\n"); defer p.deinit(); |
