diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-29 08:11:20 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-10-01 00:12:16 -0300 |
| commit | 6c10e2b37ee7b95485fb86941de32991a1732892 (patch) | |
| tree | a26e3bc3f94318b242cddd75c58f715e749f197e | |
| parent | f7e3441698a4621bec0d7c14eb45303753ffbead (diff) | |
| download | pardes-6c10e2b37ee7b95485fb86941de32991a1732892.tar.gz pardes-6c10e2b37ee7b95485fb86941de32991a1732892.zip | |
A name write is one name, refused with its reason otherwise; a log record is one line of UTF-8
A name took a second line, DEL, a C1 control or bytes that are not UTF-8,
which then went into /index and /log as they were and split or garbled a
reader's lines. `name` strips one trailing newline and refuses the rest,
saying which (`bad character in file name: not UTF-8`); and the log turns
DEL and C1 into spaces, as it did control characters, and escapes bytes
that are not UTF-8 as `\xNN`.
Co-Authored-By: Claude Opus 5.5 <[email protected]>
| -rw-r--r-- | docs/fs.md | 15 | ||||
| -rw-r--r-- | src/ninep/events.zig | 55 | ||||
| -rw-r--r-- | src/ninep/pane.zig | 44 |
3 files changed, 98 insertions, 16 deletions
@@ -507,10 +507,13 @@ writing it renames the buffer; a relative name resolves against the pane's directory. A name alone is no edit: the pane's `dirty` stays what its text made it (a renamed clean file is still 0, and nothing asks about it at Exit, Restore or Del, which ask only about text edited), and `Save` writes it -under the new name all the same. The write is the name and its newline, nothing trimmed: a blank -inside a name is taken (`two words.zig`), but one at either end, or a -control character, is refused, `bad character in file name` (EINVAL), as -acme refuses a blank (xfid.c:650), rather than quietly cut off. `body` appends on write and replaces on truncating open. A +under the new name all the same. A write is one name: one trailing newline +is its end and nothing else is trimmed. A blank inside a name is taken +(`two words.zig`); refused, EINVAL, in acme's words (xfid.c:650) and why, +are a blank at either end (not quietly cut off), `bad character in file +name: a blank at its end`, a second line, `...: a newline (a name is one +line)`, a control byte, DEL or a C1 control (U+0080-U+009F), `...: a +control character`, and bytes that are not UTF-8, `...: not UTF-8`. `body` appends on write and replaces on truncating open. A terminal's `body` is its history as plain text, frozen per open, in logical lines: a row the terminal wrapped is joined back to the row before it (the wrap is ghostty's, as `pty/run`'s output unwraps), and the last line ends @@ -786,7 +789,9 @@ acme's `errors` only takes text, and one record stream is simpler to watch than a file per pane. A `msg` said while a pane is being made can precede that pane's `new`; panes present at boot are recorded before anything else. -Control characters in a record become spaces, so a record is one line. +Control characters, DEL and C1 controls (U+0080-U+009F) in a record become +spaces, and a byte that is not UTF-8 is written `\xNN`, so a record is one +line of UTF-8. (An `event` record is not: acme's `<origin><action><q0> <q1> <flag> <n> <text>\n`, whose text may hold newlines. The origin is `E` (a 9P write to body or tag), `F` (other files, the editor's own lines), `K` (the keyboard) or `M` (the mouse); the diff --git a/src/ninep/events.zig b/src/ninep/events.zig index 98799774..d7079c5e 100644 --- a/src/ninep/events.zig +++ b/src/ninep/events.zig @@ -292,11 +292,13 @@ pub fn dropMessage(p: *Pardes, text: []const u8) void { /// counts that one instead: `<record> (x2)`, the count being every time it /// was said. One a follower has already read is not rewritten: the repeat /// is a new line carrying the running count, `(x3)`. -fn pushCounting(p: *Pardes, record: []u8) void { +fn pushCounting(p: *Pardes, raw: []u8) void { // A client retrying a write that fails the same way would fill the ring // with one line, so a repeat of the newest record is that record counted, // `(x3)`, as +Messages counts its repeats; unless a follower has read it // already and so waits on the repeat as a line of its own. + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); const last = newest(p) orelse return pushLog(p, record); var said = last.text; var times: u32 = 1; @@ -340,10 +342,47 @@ fn followerRead(p: *Pardes, seq: u64) bool { /// The log is one ring that records whether or not anyone reads it. A record /// is one line: a newline in a message or a name would read as two records. -fn pushLog(p: *Pardes, record: []u8) void { - for (record[0 .. record.len - 1]) |*c| if (c.* < ' ') { - c.* = ' '; - }; +/// A record as one line of text: a control character, DEL or a C1 control +/// (U+0080-U+009F) is a space each, and a byte that is not UTF-8 is +/// written `\xNN`, so a reader splitting on newlines and decoding UTF-8 +/// never trips. The record's own newline, last, is kept. +fn sanitize(record: []const u8, out: []u8) []u8 { + var w: usize = 0; + var i: usize = 0; + const body = record[0 .. record.len - 1]; + while (i < body.len and w + 4 < out.len) { + const c = body[i]; + if (c < ' ' or c == 0x7f) { + out[w] = ' '; + w += 1; + i += 1; + continue; + } + const n = std.unicode.utf8ByteSequenceLength(c) catch 0; + if (n == 0 or i + n > body.len or !std.unicode.utf8ValidateSlice(body[i .. i + n])) { + _ = std.fmt.bufPrint(out[w..], "\\x{x:0>2}", .{c}) catch break; + w += 4; + i += 1; + continue; + } + if (n == 2 and c == 0xC2 and body[i + 1] <= 0x9F) { + out[w] = ' '; + w += 1; + i += 2; + continue; + } + if (w + n + 1 > out.len) break; + @memcpy(out[w..][0..n], body[i..][0..n]); + w += n; + i += n; + } + out[w] = '\n'; + return out[0 .. w + 1]; +} + +fn pushLog(p: *Pardes, raw: []u8) void { + var clean: [4 * 4096 + 256]u8 = undefined; + const record = sanitize(raw, &clean); // One record larger than the ring would push every other out and then // not fit itself; cut it to what fits instead, on a character boundary. var end = @min(record.len, p.fs.log.cap - 4) - 1; @@ -1244,6 +1283,12 @@ test "a long msg record is cut between words at its cap, with an ellipsis" { _ = call(p, .{ .tag = 3, .op = .release, .node = log, .handle = h }); } +test "a record is one line of UTF-8: DEL and C1 are spaces, bytes not UTF-8 are escaped" { + var out: [64]u8 = undefined; + try testing.expectEqualStrings("msg - a b c \\xff d\n", sanitize("msg - a\x7fb\xc2\x85c \xff d\n", &out)); + try testing.expectEqualStrings("msg - caf\xc3\xa9\n", sanitize("msg - caf\xc3\xa9\n", &out)); +} + test "a long err record is cut between words, with an ellipsis" { const p = try withFile(testing.allocator, "x\n"); defer p.deinit(); diff --git a/src/ninep/pane.zig b/src/ninep/pane.zig index 5f8fc00a..66b3728c 100644 --- a/src/ninep/pane.zig +++ b/src/ninep/pane.zig @@ -597,12 +597,27 @@ fn writeFlag(p: *Pardes, req: Req, pane: *Pane, file: PaneFile) Reply { /// quietly cut off, so the name a script wrote is the name it gets. const e_name_char = "bad character in file name"; +/// Why `name` is not one file name, with the reason, or null: a newline, a +/// control byte, DEL or a C1 control (U+0080-U+009F), a blank at either +/// end, or bytes that are not UTF-8. +fn nameFault(name: []const u8) ?[]const u8 { + if (std.mem.indexOfScalar(u8, name, '\n') != null) return e_name_char ++ ": a newline (a name is one line)"; + for (name) |c| if (c < ' ' or c == 0x7f) return e_name_char ++ ": a control character"; + if (!std.unicode.utf8ValidateSlice(name)) return e_name_char ++ ": not UTF-8"; + if (std.mem.indexOf(u8, name, "\xc2") != null) { + var i: usize = 0; + while (std.mem.indexOfScalarPos(u8, name, i, 0xC2)) |at| : (i = at + 1) + if (at + 1 < name.len and name[at + 1] <= 0x9F) return e_name_char ++ ": a control character"; + } + if (name[0] == ' ' or name[name.len - 1] == ' ') return e_name_char ++ ": a blank at its end"; + return null; +} + fn writeName(p: *Pardes, req: Req, id: usize, pane: *Pane) Reply { - // One line: its newline ends it, as `echo` writes it. + // One name: its newline ends it, as `echo` writes it, and it is one. const name = if (std.mem.endsWith(u8, req.data, "\n")) req.data[0 .. req.data.len - 1] else req.data; if (name.len == 0) return Reply.fail(req.tag, E.INVAL); - for (name) |c| if (c < ' ') return tree.failText(req.tag, E.INVAL, e_name_char); - if (name[0] == ' ' or name[name.len - 1] == ' ') return tree.failText(req.tag, E.INVAL, e_name_char); + if (nameFault(name)) |why| return tree.failText(req.tag, E.INVAL, why); if (fileOf(pane) == null) return tree.failText(req.tag, E.PERM, if (pane.isTerminal()) "rename not allowed: a terminal is named by its shell's directory; cd there, or Tty in another" else @@ -1105,9 +1120,26 @@ test "name reads the file name and writing it promotes a scratch without touchin try testing.expectEqualStrings("existing target\n", target); for ([_][]const u8{ "", "\n", "bad\x01name\n" }) |bad| try testing.expectEqual(E.INVAL, wr(p, name, bad).errno()); - // A blank at either end is refused as acme refuses one, not cut off. - for ([_][]const u8{ "trailing.zig \n", " leading.zig\n", "tab\t.zig\n" }) |bad| - try testing.expectEqualStrings("bad character in file name", wr(p, name, bad).reply.ename); + // One name, in acme's words and why: a blank at either end (not cut + // off), a second line, a control byte, DEL, a C1 control, not UTF-8. + for ([_][2][]const u8{ + .{ "trailing.zig \n", "a blank at its end" }, + .{ " leading.zig\n", "a blank at its end" }, + .{ "tab\t.zig\n", "a control character" }, + .{ "two\nlines\n", "a newline" }, + .{ "del\x7f.zig\n", "a control character" }, + .{ "c1\xc2\x85.zig\n", "a control character" }, + .{ "bad\xff.zig\n", "not UTF-8" }, + }) |c| { + const refused = wr(p, name, c[0]); + try testing.expectEqual(E.INVAL, refused.errno()); + try testing.expectStringStartsWith(refused.reply.ename, "bad character in file name: "); + try testing.expect(std.mem.indexOf(u8, refused.reply.ename, c[1]) != null); + } + // A name that is one line and UTF-8 is taken, é and all. + try testing.expectEqual(Status.ok, wr(p, name, "caf\xc3\xa9.zig\n").reply.status); + try testing.expect(std.mem.endsWith(u8, pane.file.?.path, "/caf\xc3\xa9.zig")); + try testing.expectEqual(Status.ok, wr(p, name, try std.fmt.bufPrint(&line, "{s}\n", .{path})).reply.status); try testing.expectEqualStrings(path, pane.file.?.path); try testing.expectEqual(Status.ok, wr(p, name, "two words.zig\n").reply.status); const spaced = try std.fmt.bufPrint(&path_buffer, "{s}/two words.zig", .{directory}); |
