diff options
| author | Gabriel Schneider <[email protected]> | 2026-09-28 10:26:47 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-10-01 00:12:14 -0300 |
| commit | 60e40675c82efd17aa6bf31bc66c3f8f3f918360 (patch) | |
| tree | 1be23a1372c7383a2d37fe82ea6e8af9b91f488c | |
| parent | d9c6786e19567878e3c89ae212296ce941484096 (diff) | |
| download | pardes-60e40675c82efd17aa6bf31bc66c3f8f3f918360.tar.gz pardes-60e40675c82efd17aa6bf31bc66c3f8f3f918360.zip | |
Addresses search a line at a time as sam does, say why they fail, and a failed addr leaves no address
addr's regexps ran mvzr over text[from..hi]: ^ and $ anchored only at the
slice's ends, . matched newlines, the search never wrapped, and every
failure read as bad address syntax. The regexps stay mvzr's, called the
way sam searches (editors/acme/regx.c): one line per haystack, so ^ and $
fall at line boundaries and . never crosses a newline; a pattern naming
\\n runs over the whole text with its . made [^\\n]; /re/ wraps unless
limit is set, and ?re? takes the last match before the range. The ceiling
(mvzr's first alternative, not sam's longest) is documented. A failed
address says why (no match for regexp, address out of range, bad regular
expression), and a failed write to addr leaves no address, so data and
xdata refuse until the next good one instead of writing at the old range.
Co-Authored-By: Claude Opus 5.5 <[email protected]>
| -rw-r--r-- | .agents/skills/pardes-9p/SKILL.md | 8 | ||||
| -rw-r--r-- | docs/fs.md | 23 | ||||
| -rw-r--r-- | src/ninep/addr.zig | 217 | ||||
| -rw-r--r-- | src/ninep/pane.zig | 46 | ||||
| -rw-r--r-- | src/ninep/tree.zig | 2 | ||||
| -rw-r--r-- | test/fs.py | 12 |
6 files changed, 262 insertions, 46 deletions
diff --git a/.agents/skills/pardes-9p/SKILL.md b/.agents/skills/pardes-9p/SKILL.md index 7e61b73d..0207f14c 100644 --- a/.agents/skills/pardes-9p/SKILL.md +++ b/.agents/skills/pardes-9p/SKILL.md @@ -126,7 +126,13 @@ For a file or scratch pane, with `pane=$m/pane/<serial>`: `addr`, `dot` and `limit` each read the pair of offsets they also accept, which is why copying one onto another is all that acme's `addr=dot`, `dot=addr` and `limit=addr` ever were; a write may also be an address expression (`#0,#5`, -`/pattern/`, `2+1`). Moving `dot` scrolls the pane to it. `limit` bounds a +`/pattern/`, `2+1`), whose regexps are mvzr's searched as sam searches: +`^`/`$` match at any line's start and end, `.` and `[^...]` never match a +newline, the leftmost match wins (the first alternative there, not the +longest), and `/re/` wraps unless `limit` is set. A failed address says +why (`no match for regexp`, `address out of range`) and leaves no address: +`data` refuses until the next good one, so a missed target is never +written at the old one. Moving `dot` scrolls the pane to it. `limit` bounds a search and reads empty until set; truncate it to lift it. `dirty`, `mark` and `scroll` read `0` or `1` and take `0` or `1`: whether the @@ -243,6 +243,29 @@ whole buffer. until someone writes or truncates it, so writing an address and reading it back evaluates it, which is what acme(4) promises of its own `addr`. +The regular expressions are mvzr's (sets, `\d`/`\w`/`\s`, `{m,n}` and +lazy `*?` included), searched the way sam searches (editors/acme/regx.c): as +lines, so `^` and `$` match at the start and end of any line, `.` and a +negated class never match a newline, and `$` also matches at the end of a +text with no final newline. A pattern that names a newline (`\n`) runs over +the whole text instead, its `.` kept to one line. `/re/` searches forward +from the end of the current range to `limit` if one is set, and otherwise +wraps to the start of the text; `?re?` finds the last match ending before +the range, wrapping to the text's last. The match is the leftmost, but of +the alternatives at that place mvzr takes the first that matches where sam +takes the longest (`/gam|gamma/` finds `gam`); in a search begun in the +middle of a line, `^` inside an alternation can match there; and in a +pattern that spans lines, `^`, `$` and `[^...]` keep mvzr's own meaning. +pardes has no regex engine of its own on purpose; these are its limits. + +An address that does not evaluate fails the write with why: `bad address +syntax`, `no match for regexp`, `address out of range` or `bad regular +expression`. A failed write to `addr` leaves no address at all, where acme +keeps the old one: until an address is written or `addr` is truncated, +reading `addr`, and reading, writing or truncating `data` and `xdata`, fail +with `no address: the last one written to addr failed`, so a script that +missed its target cannot then write at the last one. + The three flag files `dirty`, `mark` and `scroll` read `0` or `1` and take `0` or `1`: whether the buffer differs from its file, whether a write pushes an undo point (writing `1` pushes one now), and whether a write scrolls the diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig index e41d306e..95e699e3 100644 --- a/src/ninep/addr.zig +++ b/src/ninep/addr.zig @@ -1,4 +1,5 @@ -//! The address language of a pane's `addr` file: acme's, with mvzr regexps. +//! The address language of a pane's `addr` file, acme's (editors/acme/ +//! addr.c), its regular expressions run by mvzr the way sam's run. const std = @import("std"); const mvzr = @import("mvzr"); const modal = @import("../modal.zig"); @@ -10,15 +11,10 @@ fn clip(n: usize) u32 { return std.math.cast(u32, n) orelse std.math.maxInt(u32); } -fn safePattern(pat: []const u8) bool { - var i: usize = 0; - while (i < pat.len) : (i += 1) { - if (pat[i] != '\\') continue; - if (i + 1 >= pat.len) return false; - i += 1; - } - return true; -} +pub const e_no_match = "no match for regexp"; +pub const e_range = "address out of range"; +pub const e_regexp = "bad regular expression"; +pub const e_syntax = "bad address syntax"; pub const Addr = struct { text: []const u8, @@ -26,6 +22,8 @@ pub const Addr = struct { expr: []const u8, i: usize = 0, depth: u8 = 0, + /// Why `address` answered null. + err: []const u8 = e_syntax, const max_depth = 32; const Size = enum { char, line }; @@ -134,7 +132,10 @@ pub const Addr = struct { if (r.q0 == 0 and n > 0) r.q0 = clip(a.text.len); off = @as(i64, r.q0) - n; } - if (off < 0 or off > @as(i64, @intCast(a.text.len))) return null; + if (off < 0 or off > @as(i64, @intCast(a.text.len))) { + a.err = e_range; + return null; + } const g = clip(modal.graphemeStart(a.text, @intCast(off))); return .{ .q0 = g, .q1 = g }; } @@ -154,7 +155,10 @@ pub const Addr = struct { } q0 -= 1; } - if (line > 1) return null; + if (line > 1) { + a.err = e_range; + return null; + } while (q0 > 0 and a.text[q0 - 1] != '\n') q0 -= 1; return .{ .q0 = clip(q0), .q1 = clip(q1) }; }, @@ -177,29 +181,112 @@ pub const Addr = struct { if (line > 0) q0 = q1; } } - if (line > 0) return null; + if (line > 0) { + a.err = e_range; + return null; + } return .{ .q0 = clip(q0), .q1 = clip(q1) }; } + /// acme's regexp(): forward from the end of `r` to the limit, wrapping + /// to the start of the text when there is none; backward, the last + /// match ending by the start of `r`, else the last one in the text. fn regexp(a: *Addr, r: Range, pat: []const u8, back: bool) ?Range { - if (pat.len == 0 or !safePattern(pat)) return null; - const re = mvzr.compile(pat) orelse return null; - if (back) { - const hi = @min(@as(usize, r.q0), a.text.len); - var best: ?mvzr.Match = null; - var at: usize = 0; - while (at < hi) { - const m = re.matchPos(at, a.text[0..hi]) orelse break; - best = m; - at = if (m.end > m.start) m.end else m.end + 1; + if (pat.len == 0) { + a.err = "no previous regular expression"; + return null; + } + // sam searches the text as lines: `^` and `$` at any line's start + // and end, and `.` never a newline. mvzr has no such mode (its `^` + // and `$` are the haystack's ends, its `.` any byte), so each line + // is its own haystack, and a pattern that names a newline (`\n`) + // runs over the whole text with its `.`s made `[^\n]`. + // ponytail: mvzr takes the first alternative that matches, not + // sam's longest (`/gam|gamma/` finds `gam`); a search from the middle + // of a line lets `^` match there unless the pattern starts with it; + // across lines, `^`, `$` and `[^...]` keep mvzr's meaning. A regex + // engine of sam's own would lift these; the user chose not to. + const spans = std.mem.indexOf(u8, pat, "\\n") != null; + var buf: [256]u8 = undefined; + var len: usize = 0; + var i: usize = 0; + var in_class = false; + while (i < pat.len) : (i += 1) { + const c = pat[i]; + const piece: []const u8 = if (c == '\\') piece: { + if (i + 1 >= pat.len) break :piece ""; + i += 1; + break :piece pat[i - 1 .. i + 1]; + } else if (c == '.' and spans and !in_class) "[^\\n]" else pat[i .. i + 1]; + if (piece.len == 0 or len + piece.len > buf.len) { + a.err = e_regexp; + return null; } - const m = best orelse return null; - return .{ .q0 = clip(m.start), .q1 = clip(m.end) }; + if (c == '[') in_class = true; + if (c == ']') in_class = false; + @memcpy(buf[len..][0..piece.len], piece); + len += piece.len; } - const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len; - const from = @min(@as(usize, r.q1), hi); - const m = re.match(a.text[from..hi]) orelse return null; - return .{ .q0 = clip(from + m.start), .q1 = clip(from + m.end) }; + const re = mvzr.compile(buf[0..len]) orelse { + a.err = e_regexp; + return null; + }; + const anchored = pat[0] == '^'; + const Search = struct { + /// The first match starting in `from..=last` that ends by `hi`. + fn first(rx: *const mvzr.Regex, text: []const u8, from: usize, last: usize, hi: usize, whole: bool, bol: bool) ?Range { + if (whole) { + const m = rx.matchPos(from, text[0..hi]) orelse return null; + return if (m.start <= last) .{ .q0 = clip(m.start), .q1 = clip(m.end) } else null; + } + var start = if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0; + var at = from - start; + // `^` cannot match in the middle of a line. + if (bol and at > 0) { + start = (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1; + at = 0; + } + while (start <= hi and start <= last) { + const end = std.mem.indexOfScalarPos(u8, text[0..hi], start, '\n') orelse hi; + const line = text[start..end]; + // matchPos finds nothing at a line's very end, where + // `$` or an empty pattern still match. + const hit: ?[2]usize = if (at < line.len) + (if (rx.matchPos(at, line)) |m| .{ m.start, m.end } else null) + else if (at == line.len and rx.isMatch(line[at..])) .{ at, at } else null; + if (hit) |h| { + if (start + h[0] > last) return null; + return .{ .q0 = clip(start + h[0]), .q1 = clip(start + h[1]) }; + } + if (end == hi) return null; + start = end + 1; + at = 0; + } + return null; + } + }; + const found: ?Range = if (back) found: { + var last: ?Range = null; + var before: ?Range = null; + var at: usize = 0; + while (at <= a.text.len) { + const m = Search.first(&re, a.text, at, a.text.len, a.text.len, spans, anchored) orelse break; + if (m.q1 <= r.q0) before = m; + last = m; + at = if (m.q1 > m.q0) m.q1 else m.q1 + 1; + } + break :found before orelse last; + } else found: { + const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len; + const from = @min(@as(usize, r.q1), hi); + if (Search.first(&re, a.text, from, hi, hi, spans, anchored)) |m| break :found m; + if (a.lim != null or from == 0) break :found null; + break :found Search.first(&re, a.text, 0, from - 1, hi, spans, anchored); + }; + return found orelse { + a.err = e_no_match; + return null; + }; } }; @@ -210,6 +297,69 @@ const Node = tree.Node; const E = tree.E; const Status = tree.Status; +test "regular expressions search lines as sam's do, and a search wraps" { + const p = try th.withFile(testing.allocator, "alpha beta\nbeta gamma\ngamma\n"); + defer p.deinit(); + const serial = th.serialOf(p); + const addr = Node.of(serial, .addr); + const Case = struct { from: []const u8, expr: []const u8, q0: u32, q1: u32 }; + for ([_]Case{ + // ^ and $ at every line's start and end, not only the text's + .{ .from = "#0", .expr = "/^beta/", .q0 = 11, .q1 = 15 }, + .{ .from = "#0", .expr = "/beta$/", .q0 = 6, .q1 = 10 }, + .{ .from = "#0", .expr = "/^gamma$/", .q0 = 22, .q1 = 27 }, + // . and [^...] stop at a newline; \n names one + .{ .from = "#0", .expr = "/a.*/", .q0 = 0, .q1 = 10 }, + .{ .from = "#0", .expr = "/a[^x]*/", .q0 = 0, .q1 = 10 }, + .{ .from = "#0", .expr = "/a\\nbeta/", .q0 = 9, .q1 = 15 }, + // leftmost; of the alternatives there, mvzr's first + .{ .from = "#0", .expr = "/be|beta b/", .q0 = 6, .q1 = 8 }, + .{ .from = "#0", .expr = "/gam|gamma/", .q0 = 16, .q1 = 19 }, + .{ .from = "#0", .expr = "/(al)+pha?/", .q0 = 0, .q1 = 5 }, + // `^` does not match where a search starts in the middle of a line + .{ .from = "#1", .expr = "/^/", .q0 = 11, .q1 = 11 }, + // a search past the last match wraps to the text's start + .{ .from = "$", .expr = "/alpha/", .q0 = 0, .q1 = 5 }, + // backward: the last match that ends by the start of dot + .{ .from = "#16", .expr = "?beta?", .q0 = 11, .q1 = 15 }, + .{ .from = "#0", .expr = "?gamma?", .q0 = 22, .q1 = 27 }, + }) |c| { + _ = th.wr(p, addr, c.from); + const w = th.wr(p, addr, c.expr); + try testing.expectEqual(Status.ok, w.reply.status); + try testing.expectEqual(c.q0, p.panes[0].?.fs.addr.q0); + try testing.expectEqual(c.q1, p.panes[0].?.fs.addr.q1); + } + // $ also matches at the end of text with no newline to end it + const q = try th.withFile(testing.allocator, "last line"); + defer q.deinit(); + _ = th.wr(q, Node.of(th.serialOf(q), .addr), "#0"); + try testing.expectEqual(Status.ok, th.wr(q, Node.of(th.serialOf(q), .addr), "/line$/").reply.status); + try testing.expectEqual(@as(u32, 5), q.panes[0].?.fs.addr.q0); +} + +test "a failed address leaves none, so data refuses rather than act at the last one" { + const p = try th.withFile(testing.allocator, "one\ntwo\n"); + defer p.deinit(); + const serial = th.serialOf(p); + const addr = Node.of(serial, .addr); + const data = Node.of(serial, .data); + _ = th.wr(p, addr, "#0,#3"); + try testing.expectEqualStrings(e_no_match, th.wr(p, addr, "/zzz/").reply.ename); + try testing.expectEqualStrings(pane_files.e_addr_failed, th.wr(p, data, "ONE").reply.ename); + try testing.expectEqualStrings(pane_files.e_addr_failed, th.rd(p, data, 0, 64).reply.ename); + try testing.expectEqualStrings(pane_files.e_addr_failed, th.rd(p, addr, 0, 64).reply.ename); + try testing.expectEqual(E.INVAL, th.call(p, .{ .tag = 1, .op = .setattr, .node = Node.of(serial, .xdata), .truncate = true }).errno()); + try testing.expectEqualStrings("one\ntwo\n", p.panes[0].?.file.?.content); + // A good address, or a truncated addr, gives it one again. + _ = th.wr(p, addr, "#0,#3"); + try testing.expectEqual(Status.ok, th.wr(p, data, "ONE").reply.status); + try testing.expectEqualStrings("ONE\ntwo\n", p.panes[0].?.file.?.content); + _ = th.wr(p, addr, "99"); + _ = th.call(p, .{ .tag = 1, .op = .setattr, .node = addr, .truncate = true }); + try testing.expectEqual(Status.ok, th.wr(p, data, "0").reply.status); +} + test "the address language, form by form" { const gpa = testing.allocator; const p = try th.withFile(gpa, "one\ntwo\nthree\n"); @@ -277,6 +427,15 @@ test "the address language, form by form" { _ = th.wr(p, addr, "#0"); try testing.expectEqual(E.INVAL, th.wr(p, addr, bad).errno()); } + // Each failure says which it is. + for ([_][2][]const u8{ + .{ "zzz", e_syntax }, .{ "/nomatch/", e_no_match }, .{ "99", e_range }, + .{ "#999", e_range }, .{ "/a[/", e_regexp }, .{ "/(a/", e_regexp }, + .{ "/*a/", e_regexp }, + }) |c| { + _ = th.wr(p, addr, "#0"); + try testing.expectEqualStrings(c[1], th.wr(p, addr, c[0]).reply.ename); + } const nested = "," ** 4096; _ = th.wr(p, addr, "#0"); diff --git a/src/ninep/pane.zig b/src/ninep/pane.zig index 822e0704..0a511118 100644 --- a/src/ninep/pane.zig +++ b/src/ninep/pane.zig @@ -35,6 +35,10 @@ pub const State = struct { /// register, cleared by truncating the file, so that `cp addr dot` and /// `cat addr` answer what was written. addr: Range = .{}, + /// The last address written to `addr` failed, so there is none: `data` + /// and `xdata` refuse until one is written or `addr` is truncated, + /// rather than act at the address before it, which acme would do. + addr_failed: bool = false, limit: ?Range = null, /// Opens of `event`, which hold the pane scripted. readers: u16 = 0, @@ -262,6 +266,7 @@ pub fn read(p: *Pardes, req: Req, id: usize, pane: *Pane, file: PaneFile) Reply }, .ctl => ctl.readPane(p, req, pane), .addr => addr: { + if (pf.addr_failed) break :addr tree.failText(req.tag, E.INVAL, e_addr_failed); clampAddr(pf, bodyOf(pane).len); break :addr readRange(p, req, pf.addr); }, @@ -319,6 +324,7 @@ fn readBody(p: *Pardes, req: Req, id: usize, pane: *Pane) Reply { } fn readData(req: Req, id: usize, pane: *Pane, pf: *State, stop_at_end: bool) Reply { + if (pf.addr_failed) return tree.failText(req.tag, E.INVAL, e_addr_failed); const text = bodyOf(pane); clampAddr(pf, text.len); const q0: usize = pf.addr.q0; @@ -406,6 +412,7 @@ fn writeTag(req: Req, pane: *Pane) Reply { fn writeData(p: *Pardes, req: Req, pane: *Pane) Reply { if (fileOf(pane) == null) return Reply.fail(req.tag, E.INVAL); const pf = &pane.fs; + if (pf.addr_failed) return tree.failText(req.tag, E.INVAL, e_addr_failed); clampAddr(pf, bodyOf(pane).len); const q0: usize = pf.addr.q0; const q1: usize = @max(q0, @as(usize, pf.addr.q1)); @@ -442,30 +449,34 @@ fn pairOf(text: []const u8) ?State.Range { return .{ .q0 = q0, .q1 = @max(q0, q1) }; } -/// A range file takes an address expression, or that pair of offsets. -fn rangeOf(pf: *State, text: []const u8, data: []const u8) ?State.Range { - const expr = std.mem.trimEnd(u8, data, "\n"); - if (pairOf(expr)) |r| { - const n = clip(text.len); - return .{ .q0 = @min(r.q0, n), .q1 = @min(r.q1, n) }; - } - var a: addressing.Addr = .{ .text = text, .lim = pf.limit, .expr = expr }; - const r = a.address(pf.addr) orelse return null; - return if (a.i < expr.len) null else r; -} +pub const e_addr_failed = "no address: the last one written to addr failed"; +/// A range file takes an address expression, or that pair of offsets. One +/// that does not evaluate says why: `bad address syntax`, `no match for +/// regexp`, `address out of range`, `bad regular expression`. fn writeRange(req: Req, pane: *Pane, file: PaneFile) Reply { const pf = &pane.fs; const text = bodyOf(pane); clampAddr(pf, text.len); - const r = rangeOf(pf, text, req.data) orelse return tree.failText(req.tag, E.INVAL, tree.e_bad_addr); + const expr = std.mem.trimEnd(u8, req.data, "\n"); + var a: addressing.Addr = .{ .text = text, .lim = pf.limit, .expr = expr }; + const r = if (pairOf(expr)) |pair| + State.Range{ .q0 = @min(pair.q0, clip(text.len)), .q1 = @min(pair.q1, clip(text.len)) } + else if (a.address(pf.addr)) |found| (if (a.i < expr.len) null else found) else null; + const range = r orelse { + if (file == .addr) pf.addr_failed = true; + return tree.failText(req.tag, E.INVAL, a.err); + }; switch (file) { - .addr => pf.addr = r, - .limit => pf.limit = r, + .addr => { + pf.addr = range; + pf.addr_failed = false; + }, + .limit => pf.limit = range, // Setting dot scrolls to it, which is the whole of acme's `show`. .dot => { if (fileOf(pane) == null) return Reply.fail(req.tag, E.INVAL); - setDot(pane, r); + setDot(pane, range); }, else => unreachable, } @@ -601,7 +612,10 @@ pub fn truncate(p: *Pardes, pane: *Pane, file: PaneFile) tree.Status { pane.tag.vsel.active = false; pane.tag.nsel = 0; }, - .addr => pf.addr = .{}, + .addr => { + pf.addr = .{}; + pf.addr_failed = false; + }, .limit => pf.limit = null, .dot => if (fileOf(pane) != null) setDot(pane, .{}), else => {}, diff --git a/src/ninep/tree.zig b/src/ninep/tree.zig index e9c6b8ea..3ce7e758 100644 --- a/src/ninep/tree.zig +++ b/src/ninep/tree.zig @@ -681,6 +681,8 @@ fn setattr(p: *Pardes, req: Req, target: Target) Reply { if (req.truncate) switch (target) { .pane => |t| { const id = p.paneBySerial(t.serial) orelse return Reply.fail(req.tag, E.NOENT); + if ((t.file == .data or t.file == .xdata) and p.panes[id].?.fs.addr_failed) + return failText(req.tag, E.INVAL, pane.e_addr_failed); if (pane.truncate(p, p.panes[id].?, t.file) != .ok) return Reply.fail(req.tag, E.NOMEM); }, .top => {}, @@ -344,6 +344,18 @@ def discovery(binary, embedded=False): client.write(f'/pane/{scratch}/addr', b'#0,#5') client.write(f'/pane/{scratch}/data', b'HOWDY', truncate=True) assert client.read(f'/pane/{scratch}/body') == b'HOWDY world\nsecond line\n' + # sam's regexps: ^ at any line; a miss says so and leaves no + # address, so the data write after it refuses. + client.write(f'/pane/{scratch}/addr', b'/^second/') + assert client.read(f'/pane/{scratch}/addr') == b' 12 18 ' + for path, data, why in ((f'/pane/{scratch}/addr', b'/nowhere/', 'no match for regexp'), + (f'/pane/{scratch}/data', b'LOST', 'no address')): + try: + client.write(path, data) + raise AssertionError(f'{path} took {data!r}') + except OSError as refused: + assert why in str(refused), refused + assert client.read(f'/pane/{scratch}/body') == b'HOWDY world\nsecond line\n' client.remove(f'/pane/{scratch}') print('9P discovery: listing/stat/find are inert; new, remove, look, exec, name, sel, log, ctl lock, focus, the ctl split and commands behave') |
