From 6c10e2b37ee7b95485fb86941de32991a1732892 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 29 Sep 2026 08:11:20 -0300 Subject: A name write is one name, refused with its reason otherwise; a log record is one line of UTF-8 A name took a second line, DEL, a C1 control or bytes that are not UTF-8, which then went into /index and /log as they were and split or garbled a reader's lines. `name` strips one trailing newline and refuses the rest, saying which (`bad character in file name: not UTF-8`); and the log turns DEL and C1 into spaces, as it did control characters, and escapes bytes that are not UTF-8 as `\xNN`. Co-Authored-By: Claude Opus 5.5 --- src/ninep/events.zig | 55 +++++++++++++++++++++++++++++++++++++++++++++++----- src/ninep/pane.zig | 44 +++++++++++++++++++++++++++++++++++------ 2 files changed, 88 insertions(+), 11 deletions(-) (limited to 'src') diff --git a/src/ninep/events.zig b/src/ninep/events.zig index 98799774..d7079c5e 100644 --- a/src/ninep/events.zig +++ b/src/ninep/events.zig @@ -292,11 +292,13 @@ pub fn dropMessage(p: *Pardes, text: []const u8) void { /// counts that one instead: ` (x2)`, the count being every time it /// was said. One a follower has already read is not rewritten: the repeat /// is a new line carrying the running count, `(x3)`. -fn pushCounting(p: *Pardes, record: []u8) void { +fn pushCounting(p: *Pardes, raw: []u8) void { // A client retrying a write that fails the same way would fill the ring // with one line, so a repeat of the newest record is that record counted, // `(x3)`, as +Messages counts its repeats; unless a follower has read it // already and so waits on the repeat as a line of its own. + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); const last = newest(p) orelse return pushLog(p, record); var said = last.text; var times: u32 = 1; @@ -340,10 +342,47 @@ fn followerRead(p: *Pardes, seq: u64) bool { /// The log is one ring that records whether or not anyone reads it. A record /// is one line: a newline in a message or a name would read as two records. -fn pushLog(p: *Pardes, record: []u8) void { - for (record[0 .. record.len - 1]) |*c| if (c.* < ' ') { - c.* = ' '; - }; +/// A record as one line of text: a control character, DEL or a C1 control +/// (U+0080-U+009F) is a space each, and a byte that is not UTF-8 is +/// written `\xNN`, so a reader splitting on newlines and decoding UTF-8 +/// never trips. The record's own newline, last, is kept. +fn sanitize(record: []const u8, out: []u8) []u8 { + var w: usize = 0; + var i: usize = 0; + const body = record[0 .. record.len - 1]; + while (i < body.len and w + 4 < out.len) { + const c = body[i]; + if (c < ' ' or c == 0x7f) { + out[w] = ' '; + w += 1; + i += 1; + continue; + } + const n = std.unicode.utf8ByteSequenceLength(c) catch 0; + if (n == 0 or i + n > body.len or !std.unicode.utf8ValidateSlice(body[i .. i + n])) { + _ = std.fmt.bufPrint(out[w..], "\\x{x:0>2}", .{c}) catch break; + w += 4; + i += 1; + continue; + } + if (n == 2 and c == 0xC2 and body[i + 1] <= 0x9F) { + out[w] = ' '; + w += 1; + i += 2; + continue; + } + if (w + n + 1 > out.len) break; + @memcpy(out[w..][0..n], body[i..][0..n]); + w += n; + i += n; + } + out[w] = '\n'; + return out[0 .. w + 1]; +} + +fn pushLog(p: *Pardes, raw: []u8) void { + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); // One record larger than the ring would push every other out and then // not fit itself; cut it to what fits instead, on a character boundary. var end = @min(record.len, p.fs.log.cap - 4) - 1; @@ -1244,6 +1283,12 @@ test "a long msg record is cut between words at its cap, with an ellipsis" { _ = call(p, .{ .tag = 3, .op = .release, .node = log, .handle = h }); } +test "a record is one line of UTF-8: DEL and C1 are spaces, bytes not UTF-8 are escaped" { + var out: [64]u8 = undefined; + try testing.expectEqualStrings("msg - a b c \\xff d\n", sanitize("msg - a\x7fb\xc2\x85c \xff d\n", &out)); + try testing.expectEqualStrings("msg - caf\xc3\xa9\n", sanitize("msg - caf\xc3\xa9\n", &out)); +} + test "a long err record is cut between words, with an ellipsis" { const p = try withFile(testing.allocator, "x\n"); defer p.deinit(); diff --git a/src/ninep/pane.zig b/src/ninep/pane.zig index 5f8fc00a..66b3728c 100644 --- a/src/ninep/pane.zig +++ b/src/ninep/pane.zig @@ -597,12 +597,27 @@ fn writeFlag(p: *Pardes, req: Req, pane: *Pane, file: PaneFile) Reply { /// quietly cut off, so the name a script wrote is the name it gets. const e_name_char = "bad character in file name"; +/// Why `name` is not one file name, with the reason, or null: a newline, a +/// control byte, DEL or a C1 control (U+0080-U+009F), a blank at either +/// end, or bytes that are not UTF-8. +fn nameFault(name: []const u8) ?[]const u8 { + if (std.mem.indexOfScalar(u8, name, '\n') != null) return e_name_char ++ ": a newline (a name is one line)"; + for (name) |c| if (c < ' ' or c == 0x7f) return e_name_char ++ ": a control character"; + if (!std.unicode.utf8ValidateSlice(name)) return e_name_char ++ ": not UTF-8"; + if (std.mem.indexOf(u8, name, "\xc2") != null) { + var i: usize = 0; + while (std.mem.indexOfScalarPos(u8, name, i, 0xC2)) |at| : (i = at + 1) + if (at + 1 < name.len and name[at + 1] <= 0x9F) return e_name_char ++ ": a control character"; + } + if (name[0] == ' ' or name[name.len - 1] == ' ') return e_name_char ++ ": a blank at its end"; + return null; +} + fn writeName(p: *Pardes, req: Req, id: usize, pane: *Pane) Reply { - // One line: its newline ends it, as `echo` writes it. + // One name: its newline ends it, as `echo` writes it, and it is one. const name = if (std.mem.endsWith(u8, req.data, "\n")) req.data[0 .. req.data.len - 1] else req.data; if (name.len == 0) return Reply.fail(req.tag, E.INVAL); - for (name) |c| if (c < ' ') return tree.failText(req.tag, E.INVAL, e_name_char); - if (name[0] == ' ' or name[name.len - 1] == ' ') return tree.failText(req.tag, E.INVAL, e_name_char); + if (nameFault(name)) |why| return tree.failText(req.tag, E.INVAL, why); if (fileOf(pane) == null) return tree.failText(req.tag, E.PERM, if (pane.isTerminal()) "rename not allowed: a terminal is named by its shell's directory; cd there, or Tty in another" else @@ -1105,9 +1120,26 @@ test "name reads the file name and writing it promotes a scratch without touchin try testing.expectEqualStrings("existing target\n", target); for ([_][]const u8{ "", "\n", "bad\x01name\n" }) |bad| try testing.expectEqual(E.INVAL, wr(p, name, bad).errno()); - // A blank at either end is refused as acme refuses one, not cut off. - for ([_][]const u8{ "trailing.zig \n", " leading.zig\n", "tab\t.zig\n" }) |bad| - try testing.expectEqualStrings("bad character in file name", wr(p, name, bad).reply.ename); + // One name, in acme's words and why: a blank at either end (not cut + // off), a second line, a control byte, DEL, a C1 control, not UTF-8. + for ([_][2][]const u8{ + .{ "trailing.zig \n", "a blank at its end" }, + .{ " leading.zig\n", "a blank at its end" }, + .{ "tab\t.zig\n", "a control character" }, + .{ "two\nlines\n", "a newline" }, + .{ "del\x7f.zig\n", "a control character" }, + .{ "c1\xc2\x85.zig\n", "a control character" }, + .{ "bad\xff.zig\n", "not UTF-8" }, + }) |c| { + const refused = wr(p, name, c[0]); + try testing.expectEqual(E.INVAL, refused.errno()); + try testing.expectStringStartsWith(refused.reply.ename, "bad character in file name: "); + try testing.expect(std.mem.indexOf(u8, refused.reply.ename, c[1]) != null); + } + // A name that is one line and UTF-8 is taken, é and all. + try testing.expectEqual(Status.ok, wr(p, name, "caf\xc3\xa9.zig\n").reply.status); + try testing.expect(std.mem.endsWith(u8, pane.file.?.path, "/caf\xc3\xa9.zig")); + try testing.expectEqual(Status.ok, wr(p, name, try std.fmt.bufPrint(&line, "{s}\n", .{path})).reply.status); try testing.expectEqualStrings(path, pane.file.?.path); try testing.expectEqual(Status.ok, wr(p, name, "two words.zig\n").reply.status); const spaced = try std.fmt.bufPrint(&path_buffer, "{s}/two words.zig", .{directory}); -- cgit v1.3