diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-30 23:05:57 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-10-01 00:12:17 -0300 |
| commit | d5054831d48c8cf400383ad8d842462bf2c203f8 (patch) | |
| tree | a21515fd41f5e0448547b40312a4e443e8edcd77 | |
| parent | a628330543d1c2a0b29062036f23b1c74e4ccc6e (diff) | |
| download | pardes-d5054831d48c8cf400383ad8d842462bf2c203f8.tar.gz pardes-d5054831d48c8cf400383ad8d842462bf2c203f8.zip | |
/index and the log escape a control byte, DEL or C1 in a name as \xNN, as they do a byte not UTF-8, so the name decodes to what it is
A control byte in a name read as a space, so /index could not be
decoded back to the name the name file now accepts (\x07 and a space
read the same). Every byte a name holds that is not printable UTF-8 is
now \xNN there; a message's text keeps spaces.
Co-Authored-By: Claude Opus 5.5 <[email protected]>
| -rw-r--r-- | src/ninep/ctl.zig | 7 | ||||
| -rw-r--r-- | src/ninep/events.zig | 25 |
2 files changed, 24 insertions, 8 deletions
diff --git a/src/ninep/ctl.zig b/src/ninep/ctl.zig index aa309ad1..9a1a8da1 100644 --- a/src/ninep/ctl.zig +++ b/src/ninep/ctl.zig @@ -2964,13 +2964,14 @@ test "every EINVAL a write gets says why, in its err record too; DEL is a contro try testing.expect(th.logHas(p, "an empty name")); } -test "/index shows a name as the log does: a newline in it is \\n, controls spaces, bytes not UTF-8 \\xNN" { +test "/index shows a name as the log does: a newline in it is \\n, a control byte, C1 or byte not UTF-8 \\xNN" { const p = try th.withTerm(testing.allocator); defer p.deinit(); const term = p.paneBySerial(serialOf(p)).?; - p.setCwd(term, "/tmp/two\nlines\x7f\xff"); + p.setCwd(term, "/tmp/two\nlines\x07\x7f\xc2\x85\xff"); const index = rd(p, @intFromEnum(tree.TopFile.index), 0, 4096).bytes; - try testing.expect(std.mem.indexOf(u8, index, "/tmp/two\\nlines \\xff ") != null); + // Each byte as itself, so the name decodes to what it is. + try testing.expect(std.mem.indexOf(u8, index, "/tmp/two\\nlines\\x07\\x7f\\xc2\\x85\\xff ") != null); try testing.expectEqual(@as(usize, 1), std.mem.count(u8, index, "\n")); try testing.expect(index.len > 0); try testing.expectEqual(@as(u64, 0), call(p, .{ .tag = 1, .op = .getattr, .node = @intFromEnum(tree.TopFile.index) }).reply.attr.size); diff --git a/src/ninep/events.zig b/src/ninep/events.zig index 1ffbaab9..9d59e6cc 100644 --- a/src/ninep/events.zig +++ b/src/ninep/events.zig @@ -337,9 +337,10 @@ fn sanitize(record: []const u8, out: []u8) []u8 { /// `text` as one line of UTF-8 a reader can split and decode, as /log and /// /index show names: a backslash is `\\`, a newline `\n` (a name with one -/// stays readable as what it is, not run into the next), any other control -/// character, DEL or C1 control (U+0080-U+009F) a space, and a byte that -/// is not UTF-8 `\xNN`. `out` of 4 bytes a byte of `text` holds it all. +/// stays readable as what it is, not run into the next), and any other +/// control character, DEL, C1 control (U+0080-U+009F) or byte that is not +/// UTF-8 `\xNN`, so a name read there decodes to the bytes it is. `out` of +/// 4 bytes a byte of `text` holds it all. pub fn shown(text: []const u8, out: []u8) []u8 { return shownAs(text, out, true); } @@ -365,8 +366,15 @@ fn shownAs(text: []const u8, out: []u8, escape_newline: bool) []u8 { continue; } if (c < ' ' or c == 0x7f) { - out[w] = ' '; - w += 1; + // A name's (escape_newline) is `\xNN`, so it reads back as the + // byte it was; a message's a space. + if (escape_newline) { + _ = std.fmt.bufPrint(out[w..], "\\x{x:0>2}", .{c}) catch break; + w += 4; + } else { + out[w] = ' '; + w += 1; + } i += 1; continue; } @@ -378,6 +386,13 @@ fn shownAs(text: []const u8, out: []u8, escape_newline: bool) []u8 { continue; } if (n == 2 and c == 0xC2 and body[i + 1] <= 0x9F) { + if (escape_newline) { + if (w + 8 > out.len) break; + _ = std.fmt.bufPrint(out[w..], "\\xc2\\x{x:0>2}", .{body[i + 1]}) catch break; + w += 8; + i += 2; + continue; + } out[w] = ' '; w += 1; i += 2; |
