summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--.agents/skills/pardes-9p/SKILL.md8
-rw-r--r--docs/fs.md23
-rw-r--r--src/ninep/addr.zig217
-rw-r--r--src/ninep/pane.zig46
-rw-r--r--src/ninep/tree.zig2
-rw-r--r--test/fs.py12
6 files changed, 262 insertions, 46 deletions
diff --git a/.agents/skills/pardes-9p/SKILL.md b/.agents/skills/pardes-9p/SKILL.md
index 7e61b73d..0207f14c 100644
--- a/.agents/skills/pardes-9p/SKILL.md
+++ b/.agents/skills/pardes-9p/SKILL.md
@@ -126,7 +126,13 @@ For a file or scratch pane, with `pane=$m/pane/<serial>`:
`addr`, `dot` and `limit` each read the pair of offsets they also accept, which
is why copying one onto another is all that acme's `addr=dot`, `dot=addr` and
`limit=addr` ever were; a write may also be an address expression (`#0,#5`,
-`/pattern/`, `2+1`). Moving `dot` scrolls the pane to it. `limit` bounds a
+`/pattern/`, `2+1`), whose regexps are mvzr's searched as sam searches:
+`^`/`$` match at any line's start and end, `.` and `[^...]` never match a
+newline, the leftmost match wins (the first alternative there, not the
+longest), and `/re/` wraps unless `limit` is set. A failed address says
+why (`no match for regexp`, `address out of range`) and leaves no address:
+`data` refuses until the next good one, so a missed target is never
+written at the old one. Moving `dot` scrolls the pane to it. `limit` bounds a
search and reads empty until set; truncate it to lift it.
`dirty`, `mark` and `scroll` read `0` or `1` and take `0` or `1`: whether the
diff --git a/docs/fs.md b/docs/fs.md
index 6060949b..42e4d9b8 100644
--- a/docs/fs.md
+++ b/docs/fs.md
@@ -243,6 +243,29 @@ whole buffer.
until someone writes or truncates it, so writing an address and reading it
back evaluates it, which is what acme(4) promises of its own `addr`.
+The regular expressions are mvzr's (sets, `\d`/`\w`/`\s`, `{m,n}` and
+lazy `*?` included), searched the way sam searches (editors/acme/regx.c): as
+lines, so `^` and `$` match at the start and end of any line, `.` and a
+negated class never match a newline, and `$` also matches at the end of a
+text with no final newline. A pattern that names a newline (`\n`) runs over
+the whole text instead, its `.` kept to one line. `/re/` searches forward
+from the end of the current range to `limit` if one is set, and otherwise
+wraps to the start of the text; `?re?` finds the last match ending before
+the range, wrapping to the text's last. The match is the leftmost, but of
+the alternatives at that place mvzr takes the first that matches where sam
+takes the longest (`/gam|gamma/` finds `gam`); in a search begun in the
+middle of a line, `^` inside an alternation can match there; and in a
+pattern that spans lines, `^`, `$` and `[^...]` keep mvzr's own meaning.
+pardes has no regex engine of its own on purpose; these are its limits.
+
+An address that does not evaluate fails the write with why: `bad address
+syntax`, `no match for regexp`, `address out of range` or `bad regular
+expression`. A failed write to `addr` leaves no address at all, where acme
+keeps the old one: until an address is written or `addr` is truncated,
+reading `addr`, and reading, writing or truncating `data` and `xdata`, fail
+with `no address: the last one written to addr failed`, so a script that
+missed its target cannot then write at the last one.
+
The three flag files `dirty`, `mark` and `scroll` read `0` or `1` and take
`0` or `1`: whether the buffer differs from its file, whether a write pushes
an undo point (writing `1` pushes one now), and whether a write scrolls the
diff --git a/src/ninep/addr.zig b/src/ninep/addr.zig
index e41d306e..95e699e3 100644
--- a/src/ninep/addr.zig
+++ b/src/ninep/addr.zig
@@ -1,4 +1,5 @@
-//! The address language of a pane's `addr` file: acme's, with mvzr regexps.
+//! The address language of a pane's `addr` file, acme's (editors/acme/
+//! addr.c), its regular expressions run by mvzr the way sam's run.
const std = @import("std");
const mvzr = @import("mvzr");
const modal = @import("../modal.zig");
@@ -10,15 +11,10 @@ fn clip(n: usize) u32 {
return std.math.cast(u32, n) orelse std.math.maxInt(u32);
}
-fn safePattern(pat: []const u8) bool {
- var i: usize = 0;
- while (i < pat.len) : (i += 1) {
- if (pat[i] != '\\') continue;
- if (i + 1 >= pat.len) return false;
- i += 1;
- }
- return true;
-}
+pub const e_no_match = "no match for regexp";
+pub const e_range = "address out of range";
+pub const e_regexp = "bad regular expression";
+pub const e_syntax = "bad address syntax";
pub const Addr = struct {
text: []const u8,
@@ -26,6 +22,8 @@ pub const Addr = struct {
expr: []const u8,
i: usize = 0,
depth: u8 = 0,
+ /// Why `address` answered null.
+ err: []const u8 = e_syntax,
const max_depth = 32;
const Size = enum { char, line };
@@ -134,7 +132,10 @@ pub const Addr = struct {
if (r.q0 == 0 and n > 0) r.q0 = clip(a.text.len);
off = @as(i64, r.q0) - n;
}
- if (off < 0 or off > @as(i64, @intCast(a.text.len))) return null;
+ if (off < 0 or off > @as(i64, @intCast(a.text.len))) {
+ a.err = e_range;
+ return null;
+ }
const g = clip(modal.graphemeStart(a.text, @intCast(off)));
return .{ .q0 = g, .q1 = g };
}
@@ -154,7 +155,10 @@ pub const Addr = struct {
}
q0 -= 1;
}
- if (line > 1) return null;
+ if (line > 1) {
+ a.err = e_range;
+ return null;
+ }
while (q0 > 0 and a.text[q0 - 1] != '\n') q0 -= 1;
return .{ .q0 = clip(q0), .q1 = clip(q1) };
},
@@ -177,29 +181,112 @@ pub const Addr = struct {
if (line > 0) q0 = q1;
}
}
- if (line > 0) return null;
+ if (line > 0) {
+ a.err = e_range;
+ return null;
+ }
return .{ .q0 = clip(q0), .q1 = clip(q1) };
}
+ /// acme's regexp(): forward from the end of `r` to the limit, wrapping
+ /// to the start of the text when there is none; backward, the last
+ /// match ending by the start of `r`, else the last one in the text.
fn regexp(a: *Addr, r: Range, pat: []const u8, back: bool) ?Range {
- if (pat.len == 0 or !safePattern(pat)) return null;
- const re = mvzr.compile(pat) orelse return null;
- if (back) {
- const hi = @min(@as(usize, r.q0), a.text.len);
- var best: ?mvzr.Match = null;
- var at: usize = 0;
- while (at < hi) {
- const m = re.matchPos(at, a.text[0..hi]) orelse break;
- best = m;
- at = if (m.end > m.start) m.end else m.end + 1;
+ if (pat.len == 0) {
+ a.err = "no previous regular expression";
+ return null;
+ }
+ // sam searches the text as lines: `^` and `$` at any line's start
+ // and end, and `.` never a newline. mvzr has no such mode (its `^`
+ // and `$` are the haystack's ends, its `.` any byte), so each line
+ // is its own haystack, and a pattern that names a newline (`\n`)
+ // runs over the whole text with its `.`s made `[^\n]`.
+ // ponytail: mvzr takes the first alternative that matches, not
+ // sam's longest (`/gam|gamma/` finds `gam`); a search from the middle
+ // of a line lets `^` match there unless the pattern starts with it;
+ // across lines, `^`, `$` and `[^...]` keep mvzr's meaning. A regex
+ // engine of sam's own would lift these; the user chose not to.
+ const spans = std.mem.indexOf(u8, pat, "\\n") != null;
+ var buf: [256]u8 = undefined;
+ var len: usize = 0;
+ var i: usize = 0;
+ var in_class = false;
+ while (i < pat.len) : (i += 1) {
+ const c = pat[i];
+ const piece: []const u8 = if (c == '\\') piece: {
+ if (i + 1 >= pat.len) break :piece "";
+ i += 1;
+ break :piece pat[i - 1 .. i + 1];
+ } else if (c == '.' and spans and !in_class) "[^\\n]" else pat[i .. i + 1];
+ if (piece.len == 0 or len + piece.len > buf.len) {
+ a.err = e_regexp;
+ return null;
}
- const m = best orelse return null;
- return .{ .q0 = clip(m.start), .q1 = clip(m.end) };
+ if (c == '[') in_class = true;
+ if (c == ']') in_class = false;
+ @memcpy(buf[len..][0..piece.len], piece);
+ len += piece.len;
}
- const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len;
- const from = @min(@as(usize, r.q1), hi);
- const m = re.match(a.text[from..hi]) orelse return null;
- return .{ .q0 = clip(from + m.start), .q1 = clip(from + m.end) };
+ const re = mvzr.compile(buf[0..len]) orelse {
+ a.err = e_regexp;
+ return null;
+ };
+ const anchored = pat[0] == '^';
+ const Search = struct {
+ /// The first match starting in `from..=last` that ends by `hi`.
+ fn first(rx: *const mvzr.Regex, text: []const u8, from: usize, last: usize, hi: usize, whole: bool, bol: bool) ?Range {
+ if (whole) {
+ const m = rx.matchPos(from, text[0..hi]) orelse return null;
+ return if (m.start <= last) .{ .q0 = clip(m.start), .q1 = clip(m.end) } else null;
+ }
+ var start = if (std.mem.lastIndexOfScalar(u8, text[0..from], '\n')) |nl| nl + 1 else 0;
+ var at = from - start;
+ // `^` cannot match in the middle of a line.
+ if (bol and at > 0) {
+ start = (std.mem.indexOfScalarPos(u8, text[0..hi], from, '\n') orelse return null) + 1;
+ at = 0;
+ }
+ while (start <= hi and start <= last) {
+ const end = std.mem.indexOfScalarPos(u8, text[0..hi], start, '\n') orelse hi;
+ const line = text[start..end];
+ // matchPos finds nothing at a line's very end, where
+ // `$` or an empty pattern still match.
+ const hit: ?[2]usize = if (at < line.len)
+ (if (rx.matchPos(at, line)) |m| .{ m.start, m.end } else null)
+ else if (at == line.len and rx.isMatch(line[at..])) .{ at, at } else null;
+ if (hit) |h| {
+ if (start + h[0] > last) return null;
+ return .{ .q0 = clip(start + h[0]), .q1 = clip(start + h[1]) };
+ }
+ if (end == hi) return null;
+ start = end + 1;
+ at = 0;
+ }
+ return null;
+ }
+ };
+ const found: ?Range = if (back) found: {
+ var last: ?Range = null;
+ var before: ?Range = null;
+ var at: usize = 0;
+ while (at <= a.text.len) {
+ const m = Search.first(&re, a.text, at, a.text.len, a.text.len, spans, anchored) orelse break;
+ if (m.q1 <= r.q0) before = m;
+ last = m;
+ at = if (m.q1 > m.q0) m.q1 else m.q1 + 1;
+ }
+ break :found before orelse last;
+ } else found: {
+ const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len;
+ const from = @min(@as(usize, r.q1), hi);
+ if (Search.first(&re, a.text, from, hi, hi, spans, anchored)) |m| break :found m;
+ if (a.lim != null or from == 0) break :found null;
+ break :found Search.first(&re, a.text, 0, from - 1, hi, spans, anchored);
+ };
+ return found orelse {
+ a.err = e_no_match;
+ return null;
+ };
}
};
@@ -210,6 +297,69 @@ const Node = tree.Node;
const E = tree.E;
const Status = tree.Status;
+test "regular expressions search lines as sam's do, and a search wraps" {
+ const p = try th.withFile(testing.allocator, "alpha beta\nbeta gamma\ngamma\n");
+ defer p.deinit();
+ const serial = th.serialOf(p);
+ const addr = Node.of(serial, .addr);
+ const Case = struct { from: []const u8, expr: []const u8, q0: u32, q1: u32 };
+ for ([_]Case{
+ // ^ and $ at every line's start and end, not only the text's
+ .{ .from = "#0", .expr = "/^beta/", .q0 = 11, .q1 = 15 },
+ .{ .from = "#0", .expr = "/beta$/", .q0 = 6, .q1 = 10 },
+ .{ .from = "#0", .expr = "/^gamma$/", .q0 = 22, .q1 = 27 },
+ // . and [^...] stop at a newline; \n names one
+ .{ .from = "#0", .expr = "/a.*/", .q0 = 0, .q1 = 10 },
+ .{ .from = "#0", .expr = "/a[^x]*/", .q0 = 0, .q1 = 10 },
+ .{ .from = "#0", .expr = "/a\\nbeta/", .q0 = 9, .q1 = 15 },
+ // leftmost; of the alternatives there, mvzr's first
+ .{ .from = "#0", .expr = "/be|beta b/", .q0 = 6, .q1 = 8 },
+ .{ .from = "#0", .expr = "/gam|gamma/", .q0 = 16, .q1 = 19 },
+ .{ .from = "#0", .expr = "/(al)+pha?/", .q0 = 0, .q1 = 5 },
+ // `^` does not match where a search starts in the middle of a line
+ .{ .from = "#1", .expr = "/^/", .q0 = 11, .q1 = 11 },
+ // a search past the last match wraps to the text's start
+ .{ .from = "$", .expr = "/alpha/", .q0 = 0, .q1 = 5 },
+ // backward: the last match that ends by the start of dot
+ .{ .from = "#16", .expr = "?beta?", .q0 = 11, .q1 = 15 },
+ .{ .from = "#0", .expr = "?gamma?", .q0 = 22, .q1 = 27 },
+ }) |c| {
+ _ = th.wr(p, addr, c.from);
+ const w = th.wr(p, addr, c.expr);
+ try testing.expectEqual(Status.ok, w.reply.status);
+ try testing.expectEqual(c.q0, p.panes[0].?.fs.addr.q0);
+ try testing.expectEqual(c.q1, p.panes[0].?.fs.addr.q1);
+ }
+ // $ also matches at the end of text with no newline to end it
+ const q = try th.withFile(testing.allocator, "last line");
+ defer q.deinit();
+ _ = th.wr(q, Node.of(th.serialOf(q), .addr), "#0");
+ try testing.expectEqual(Status.ok, th.wr(q, Node.of(th.serialOf(q), .addr), "/line$/").reply.status);
+ try testing.expectEqual(@as(u32, 5), q.panes[0].?.fs.addr.q0);
+}
+
+test "a failed address leaves none, so data refuses rather than act at the last one" {
+ const p = try th.withFile(testing.allocator, "one\ntwo\n");
+ defer p.deinit();
+ const serial = th.serialOf(p);
+ const addr = Node.of(serial, .addr);
+ const data = Node.of(serial, .data);
+ _ = th.wr(p, addr, "#0,#3");
+ try testing.expectEqualStrings(e_no_match, th.wr(p, addr, "/zzz/").reply.ename);
+ try testing.expectEqualStrings(pane_files.e_addr_failed, th.wr(p, data, "ONE").reply.ename);
+ try testing.expectEqualStrings(pane_files.e_addr_failed, th.rd(p, data, 0, 64).reply.ename);
+ try testing.expectEqualStrings(pane_files.e_addr_failed, th.rd(p, addr, 0, 64).reply.ename);
+ try testing.expectEqual(E.INVAL, th.call(p, .{ .tag = 1, .op = .setattr, .node = Node.of(serial, .xdata), .truncate = true }).errno());
+ try testing.expectEqualStrings("one\ntwo\n", p.panes[0].?.file.?.content);
+ // A good address, or a truncated addr, gives it one again.
+ _ = th.wr(p, addr, "#0,#3");
+ try testing.expectEqual(Status.ok, th.wr(p, data, "ONE").reply.status);
+ try testing.expectEqualStrings("ONE\ntwo\n", p.panes[0].?.file.?.content);
+ _ = th.wr(p, addr, "99");
+ _ = th.call(p, .{ .tag = 1, .op = .setattr, .node = addr, .truncate = true });
+ try testing.expectEqual(Status.ok, th.wr(p, data, "0").reply.status);
+}
+
test "the address language, form by form" {
const gpa = testing.allocator;
const p = try th.withFile(gpa, "one\ntwo\nthree\n");
@@ -277,6 +427,15 @@ test "the address language, form by form" {
_ = th.wr(p, addr, "#0");
try testing.expectEqual(E.INVAL, th.wr(p, addr, bad).errno());
}
+ // Each failure says which it is.
+ for ([_][2][]const u8{
+ .{ "zzz", e_syntax }, .{ "/nomatch/", e_no_match }, .{ "99", e_range },
+ .{ "#999", e_range }, .{ "/a[/", e_regexp }, .{ "/(a/", e_regexp },
+ .{ "/*a/", e_regexp },
+ }) |c| {
+ _ = th.wr(p, addr, "#0");
+ try testing.expectEqualStrings(c[1], th.wr(p, addr, c[0]).reply.ename);
+ }
const nested = "," ** 4096;
_ = th.wr(p, addr, "#0");
diff --git a/src/ninep/pane.zig b/src/ninep/pane.zig
index 822e0704..0a511118 100644
--- a/src/ninep/pane.zig
+++ b/src/ninep/pane.zig
@@ -35,6 +35,10 @@ pub const State = struct {
/// register, cleared by truncating the file, so that `cp addr dot` and
/// `cat addr` answer what was written.
addr: Range = .{},
+ /// The last address written to `addr` failed, so there is none: `data`
+ /// and `xdata` refuse until one is written or `addr` is truncated,
+ /// rather than act at the address before it, which acme would do.
+ addr_failed: bool = false,
limit: ?Range = null,
/// Opens of `event`, which hold the pane scripted.
readers: u16 = 0,
@@ -262,6 +266,7 @@ pub fn read(p: *Pardes, req: Req, id: usize, pane: *Pane, file: PaneFile) Reply
},
.ctl => ctl.readPane(p, req, pane),
.addr => addr: {
+ if (pf.addr_failed) break :addr tree.failText(req.tag, E.INVAL, e_addr_failed);
clampAddr(pf, bodyOf(pane).len);
break :addr readRange(p, req, pf.addr);
},
@@ -319,6 +324,7 @@ fn readBody(p: *Pardes, req: Req, id: usize, pane: *Pane) Reply {
}
fn readData(req: Req, id: usize, pane: *Pane, pf: *State, stop_at_end: bool) Reply {
+ if (pf.addr_failed) return tree.failText(req.tag, E.INVAL, e_addr_failed);
const text = bodyOf(pane);
clampAddr(pf, text.len);
const q0: usize = pf.addr.q0;
@@ -406,6 +412,7 @@ fn writeTag(req: Req, pane: *Pane) Reply {
fn writeData(p: *Pardes, req: Req, pane: *Pane) Reply {
if (fileOf(pane) == null) return Reply.fail(req.tag, E.INVAL);
const pf = &pane.fs;
+ if (pf.addr_failed) return tree.failText(req.tag, E.INVAL, e_addr_failed);
clampAddr(pf, bodyOf(pane).len);
const q0: usize = pf.addr.q0;
const q1: usize = @max(q0, @as(usize, pf.addr.q1));
@@ -442,30 +449,34 @@ fn pairOf(text: []const u8) ?State.Range {
return .{ .q0 = q0, .q1 = @max(q0, q1) };
}
-/// A range file takes an address expression, or that pair of offsets.
-fn rangeOf(pf: *State, text: []const u8, data: []const u8) ?State.Range {
- const expr = std.mem.trimEnd(u8, data, "\n");
- if (pairOf(expr)) |r| {
- const n = clip(text.len);
- return .{ .q0 = @min(r.q0, n), .q1 = @min(r.q1, n) };
- }
- var a: addressing.Addr = .{ .text = text, .lim = pf.limit, .expr = expr };
- const r = a.address(pf.addr) orelse return null;
- return if (a.i < expr.len) null else r;
-}
+pub const e_addr_failed = "no address: the last one written to addr failed";
+/// A range file takes an address expression, or that pair of offsets. One
+/// that does not evaluate says why: `bad address syntax`, `no match for
+/// regexp`, `address out of range`, `bad regular expression`.
fn writeRange(req: Req, pane: *Pane, file: PaneFile) Reply {
const pf = &pane.fs;
const text = bodyOf(pane);
clampAddr(pf, text.len);
- const r = rangeOf(pf, text, req.data) orelse return tree.failText(req.tag, E.INVAL, tree.e_bad_addr);
+ const expr = std.mem.trimEnd(u8, req.data, "\n");
+ var a: addressing.Addr = .{ .text = text, .lim = pf.limit, .expr = expr };
+ const r = if (pairOf(expr)) |pair|
+ State.Range{ .q0 = @min(pair.q0, clip(text.len)), .q1 = @min(pair.q1, clip(text.len)) }
+ else if (a.address(pf.addr)) |found| (if (a.i < expr.len) null else found) else null;
+ const range = r orelse {
+ if (file == .addr) pf.addr_failed = true;
+ return tree.failText(req.tag, E.INVAL, a.err);
+ };
switch (file) {
- .addr => pf.addr = r,
- .limit => pf.limit = r,
+ .addr => {
+ pf.addr = range;
+ pf.addr_failed = false;
+ },
+ .limit => pf.limit = range,
// Setting dot scrolls to it, which is the whole of acme's `show`.
.dot => {
if (fileOf(pane) == null) return Reply.fail(req.tag, E.INVAL);
- setDot(pane, r);
+ setDot(pane, range);
},
else => unreachable,
}
@@ -601,7 +612,10 @@ pub fn truncate(p: *Pardes, pane: *Pane, file: PaneFile) tree.Status {
pane.tag.vsel.active = false;
pane.tag.nsel = 0;
},
- .addr => pf.addr = .{},
+ .addr => {
+ pf.addr = .{};
+ pf.addr_failed = false;
+ },
.limit => pf.limit = null,
.dot => if (fileOf(pane) != null) setDot(pane, .{}),
else => {},
diff --git a/src/ninep/tree.zig b/src/ninep/tree.zig
index e9c6b8ea..3ce7e758 100644
--- a/src/ninep/tree.zig
+++ b/src/ninep/tree.zig
@@ -681,6 +681,8 @@ fn setattr(p: *Pardes, req: Req, target: Target) Reply {
if (req.truncate) switch (target) {
.pane => |t| {
const id = p.paneBySerial(t.serial) orelse return Reply.fail(req.tag, E.NOENT);
+ if ((t.file == .data or t.file == .xdata) and p.panes[id].?.fs.addr_failed)
+ return failText(req.tag, E.INVAL, pane.e_addr_failed);
if (pane.truncate(p, p.panes[id].?, t.file) != .ok) return Reply.fail(req.tag, E.NOMEM);
},
.top => {},
diff --git a/test/fs.py b/test/fs.py
index ca1ca70e..a6d24e0b 100644
--- a/test/fs.py
+++ b/test/fs.py
@@ -344,6 +344,18 @@ def discovery(binary, embedded=False):
client.write(f'/pane/{scratch}/addr', b'#0,#5')
client.write(f'/pane/{scratch}/data', b'HOWDY', truncate=True)
assert client.read(f'/pane/{scratch}/body') == b'HOWDY world\nsecond line\n'
+ # sam's regexps: ^ at any line; a miss says so and leaves no
+ # address, so the data write after it refuses.
+ client.write(f'/pane/{scratch}/addr', b'/^second/')
+ assert client.read(f'/pane/{scratch}/addr') == b' 12 18 '
+ for path, data, why in ((f'/pane/{scratch}/addr', b'/nowhere/', 'no match for regexp'),
+ (f'/pane/{scratch}/data', b'LOST', 'no address')):
+ try:
+ client.write(path, data)
+ raise AssertionError(f'{path} took {data!r}')
+ except OSError as refused:
+ assert why in str(refused), refused
+ assert client.read(f'/pane/{scratch}/body') == b'HOWDY world\nsecond line\n'
client.remove(f'/pane/{scratch}')
print('9P discovery: listing/stat/find are inert; new, remove, look, exec, name, sel, log, ctl lock, focus, the ctl split and commands behave')