summaryrefslogtreecommitdiff
path: root/src/ninep
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-29 08:11:20 -0300
committerGabriel Schneider <[email protected]>2026-10-01 00:12:16 -0300
commit6c10e2b37ee7b95485fb86941de32991a1732892 (patch)
treea26e3bc3f94318b242cddd75c58f715e749f197e /src/ninep
parentf7e3441698a4621bec0d7c14eb45303753ffbead (diff)
downloadpardes-6c10e2b37ee7b95485fb86941de32991a1732892.tar.gz
pardes-6c10e2b37ee7b95485fb86941de32991a1732892.zip
A name write is one name, refused with its reason otherwise; a log record is one line of UTF-8
A name took a second line, DEL, a C1 control or bytes that are not UTF-8, which then went into /index and /log as they were and split or garbled a reader's lines. `name` strips one trailing newline and refuses the rest, saying which (`bad character in file name: not UTF-8`); and the log turns DEL and C1 into spaces, as it did control characters, and escapes bytes that are not UTF-8 as `\xNN`. Co-Authored-By: Claude Opus 5.5 <[email protected]>
Diffstat (limited to 'src/ninep')
-rw-r--r--src/ninep/events.zig55
-rw-r--r--src/ninep/pane.zig44
2 files changed, 88 insertions, 11 deletions
diff --git a/src/ninep/events.zig b/src/ninep/events.zig
index 98799774..d7079c5e 100644
--- a/src/ninep/events.zig
+++ b/src/ninep/events.zig
@@ -292,11 +292,13 @@ pub fn dropMessage(p: *Pardes, text: []const u8) void {
/// counts that one instead: `<record> (x2)`, the count being every time it
/// was said. One a follower has already read is not rewritten: the repeat
/// is a new line carrying the running count, `(x3)`.
-fn pushCounting(p: *Pardes, record: []u8) void {
+fn pushCounting(p: *Pardes, raw: []u8) void {
// A client retrying a write that fails the same way would fill the ring
// with one line, so a repeat of the newest record is that record counted,
// `(x3)`, as +Messages counts its repeats; unless a follower has read it
// already and so waits on the repeat as a line of its own.
+ var clean: [4 * 4096 + 256]u8 = undefined;
+ const record = sanitize(raw, &clean);
const last = newest(p) orelse return pushLog(p, record);
var said = last.text;
var times: u32 = 1;
@@ -340,10 +342,47 @@ fn followerRead(p: *Pardes, seq: u64) bool {
/// The log is one ring that records whether or not anyone reads it. A record
/// is one line: a newline in a message or a name would read as two records.
-fn pushLog(p: *Pardes, record: []u8) void {
- for (record[0 .. record.len - 1]) |*c| if (c.* < ' ') {
- c.* = ' ';
- };
+/// A record as one line of text: a control character, DEL or a C1 control
+/// (U+0080-U+009F) is a space each, and a byte that is not UTF-8 is
+/// written `\xNN`, so a reader splitting on newlines and decoding UTF-8
+/// never trips. The record's own newline, last, is kept.
+fn sanitize(record: []const u8, out: []u8) []u8 {
+ var w: usize = 0;
+ var i: usize = 0;
+ const body = record[0 .. record.len - 1];
+ while (i < body.len and w + 4 < out.len) {
+ const c = body[i];
+ if (c < ' ' or c == 0x7f) {
+ out[w] = ' ';
+ w += 1;
+ i += 1;
+ continue;
+ }
+ const n = std.unicode.utf8ByteSequenceLength(c) catch 0;
+ if (n == 0 or i + n > body.len or !std.unicode.utf8ValidateSlice(body[i .. i + n])) {
+ _ = std.fmt.bufPrint(out[w..], "\\x{x:0>2}", .{c}) catch break;
+ w += 4;
+ i += 1;
+ continue;
+ }
+ if (n == 2 and c == 0xC2 and body[i + 1] <= 0x9F) {
+ out[w] = ' ';
+ w += 1;
+ i += 2;
+ continue;
+ }
+ if (w + n + 1 > out.len) break;
+ @memcpy(out[w..][0..n], body[i..][0..n]);
+ w += n;
+ i += n;
+ }
+ out[w] = '\n';
+ return out[0 .. w + 1];
+}
+
+fn pushLog(p: *Pardes, raw: []u8) void {
+ var clean: [4 * 4096 + 256]u8 = undefined;
+ const record = sanitize(raw, &clean);
// One record larger than the ring would push every other out and then
// not fit itself; cut it to what fits instead, on a character boundary.
var end = @min(record.len, p.fs.log.cap - 4) - 1;
@@ -1244,6 +1283,12 @@ test "a long msg record is cut between words at its cap, with an ellipsis" {
_ = call(p, .{ .tag = 3, .op = .release, .node = log, .handle = h });
}
+test "a record is one line of UTF-8: DEL and C1 are spaces, bytes not UTF-8 are escaped" {
+ var out: [64]u8 = undefined;
+ try testing.expectEqualStrings("msg - a b c \\xff d\n", sanitize("msg - a\x7fb\xc2\x85c \xff d\n", &out));
+ try testing.expectEqualStrings("msg - caf\xc3\xa9\n", sanitize("msg - caf\xc3\xa9\n", &out));
+}
+
test "a long err record is cut between words, with an ellipsis" {
const p = try withFile(testing.allocator, "x\n");
defer p.deinit();
diff --git a/src/ninep/pane.zig b/src/ninep/pane.zig
index 5f8fc00a..66b3728c 100644
--- a/src/ninep/pane.zig
+++ b/src/ninep/pane.zig
@@ -597,12 +597,27 @@ fn writeFlag(p: *Pardes, req: Req, pane: *Pane, file: PaneFile) Reply {
/// quietly cut off, so the name a script wrote is the name it gets.
const e_name_char = "bad character in file name";
+/// Why `name` is not one file name, with the reason, or null: a newline, a
+/// control byte, DEL or a C1 control (U+0080-U+009F), a blank at either
+/// end, or bytes that are not UTF-8.
+fn nameFault(name: []const u8) ?[]const u8 {
+ if (std.mem.indexOfScalar(u8, name, '\n') != null) return e_name_char ++ ": a newline (a name is one line)";
+ for (name) |c| if (c < ' ' or c == 0x7f) return e_name_char ++ ": a control character";
+ if (!std.unicode.utf8ValidateSlice(name)) return e_name_char ++ ": not UTF-8";
+ if (std.mem.indexOf(u8, name, "\xc2") != null) {
+ var i: usize = 0;
+ while (std.mem.indexOfScalarPos(u8, name, i, 0xC2)) |at| : (i = at + 1)
+ if (at + 1 < name.len and name[at + 1] <= 0x9F) return e_name_char ++ ": a control character";
+ }
+ if (name[0] == ' ' or name[name.len - 1] == ' ') return e_name_char ++ ": a blank at its end";
+ return null;
+}
+
fn writeName(p: *Pardes, req: Req, id: usize, pane: *Pane) Reply {
- // One line: its newline ends it, as `echo` writes it.
+ // One name: its newline ends it, as `echo` writes it, and it is one.
const name = if (std.mem.endsWith(u8, req.data, "\n")) req.data[0 .. req.data.len - 1] else req.data;
if (name.len == 0) return Reply.fail(req.tag, E.INVAL);
- for (name) |c| if (c < ' ') return tree.failText(req.tag, E.INVAL, e_name_char);
- if (name[0] == ' ' or name[name.len - 1] == ' ') return tree.failText(req.tag, E.INVAL, e_name_char);
+ if (nameFault(name)) |why| return tree.failText(req.tag, E.INVAL, why);
if (fileOf(pane) == null) return tree.failText(req.tag, E.PERM, if (pane.isTerminal())
"rename not allowed: a terminal is named by its shell's directory; cd there, or Tty in another"
else
@@ -1105,9 +1120,26 @@ test "name reads the file name and writing it promotes a scratch without touchin
try testing.expectEqualStrings("existing target\n", target);
for ([_][]const u8{ "", "\n", "bad\x01name\n" }) |bad|
try testing.expectEqual(E.INVAL, wr(p, name, bad).errno());
- // A blank at either end is refused as acme refuses one, not cut off.
- for ([_][]const u8{ "trailing.zig \n", " leading.zig\n", "tab\t.zig\n" }) |bad|
- try testing.expectEqualStrings("bad character in file name", wr(p, name, bad).reply.ename);
+ // One name, in acme's words and why: a blank at either end (not cut
+ // off), a second line, a control byte, DEL, a C1 control, not UTF-8.
+ for ([_][2][]const u8{
+ .{ "trailing.zig \n", "a blank at its end" },
+ .{ " leading.zig\n", "a blank at its end" },
+ .{ "tab\t.zig\n", "a control character" },
+ .{ "two\nlines\n", "a newline" },
+ .{ "del\x7f.zig\n", "a control character" },
+ .{ "c1\xc2\x85.zig\n", "a control character" },
+ .{ "bad\xff.zig\n", "not UTF-8" },
+ }) |c| {
+ const refused = wr(p, name, c[0]);
+ try testing.expectEqual(E.INVAL, refused.errno());
+ try testing.expectStringStartsWith(refused.reply.ename, "bad character in file name: ");
+ try testing.expect(std.mem.indexOf(u8, refused.reply.ename, c[1]) != null);
+ }
+ // A name that is one line and UTF-8 is taken, é and all.
+ try testing.expectEqual(Status.ok, wr(p, name, "caf\xc3\xa9.zig\n").reply.status);
+ try testing.expect(std.mem.endsWith(u8, pane.file.?.path, "/caf\xc3\xa9.zig"));
+ try testing.expectEqual(Status.ok, wr(p, name, try std.fmt.bufPrint(&line, "{s}\n", .{path})).reply.status);
try testing.expectEqualStrings(path, pane.file.?.path);
try testing.expectEqual(Status.ok, wr(p, name, "two words.zig\n").reply.status);
const spaced = try std.fmt.bufPrint(&path_buffer, "{s}/two words.zig", .{directory});