//! The address language of a pane's `addr` file, acme's (editors/acme/ //! addr.c), its regular expressions searched as sam searches (regexp.zig). const std = @import("std"); const regexp_ = @import("../regexp.zig"); const modal = @import("../modal.zig"); const pane_files = @import("pane.zig"); pub const Range = pane_files.State.Range; fn clip(n: usize) u32 { return std.math.cast(u32, n) orelse std.math.maxInt(u32); } pub const e_no_match = "no match for regexp"; pub const e_range = "address out of range"; pub const e_regexp = "bad regular expression"; pub const e_slow = "regular expression search took too long"; pub const e_syntax = "bad address syntax"; pub const Addr = struct { text: []const u8, lim: ?Range, expr: []const u8, i: usize = 0, depth: u8 = 0, /// Why `address` answered null. err: []const u8 = e_syntax, const max_depth = 32; const Size = enum { char, line }; pub fn address(a: *Addr, ar_in: Range) ?Range { const start = a.i; var ar = ar_in; var r = ar_in; var dir: u8 = 0; var size: Size = .line; var c: u8 = 0; while (a.i < a.expr.len) { const prevc = c; c = a.expr[a.i]; a.i += 1; switch (c) { ',', ';' => { if (c == ';') ar = r; if (prevc == 0) r.q0 = 0; // lhs defaults to 0 if (a.i >= a.expr.len) { r.q1 = clip(a.text.len); // rhs defaults to $ } else { if (a.depth >= max_depth) return null; a.depth += 1; const nr = a.address(ar) orelse return null; a.depth -= 1; r.q1 = nr.q1; } return r; }, '+', '-' => { if (prevc == '+' or prevc == '-') { const nc = if (a.i < a.expr.len) a.expr[a.i] else 0; if (nc != '#' and nc != '/' and nc != '?') r = a.number(r, 1, prevc, .line) orelse return null; } dir = c; }, '.', '$' => { if (a.i != start + 1) { a.i -= 1; return r; } r = if (c == '.') ar else .{ .q0 = clip(a.text.len), .q1 = clip(a.text.len) }; dir = if (a.i < a.expr.len) '+' else 0; }, '#', '0'...'9' => { var digit = c; if (c == '#') { if (a.i >= a.expr.len or a.expr[a.i] < '0' or a.expr[a.i] > '9') { a.i -= 1; return r; } digit = a.expr[a.i]; a.i += 1; size = .char; } var n: u64 = digit - '0'; while (a.i < a.expr.len) : (a.i += 1) { const d = a.expr[a.i]; if (d < '0' or d > '9') break; n = @min(n * 10 + (d - '0'), std.math.maxInt(u32)); } r = a.number(r, @intCast(n), dir, size) orelse return null; dir = 0; size = .line; }, '/', '?' => { const back = c == '?'; r = a.regexp(r, a.pattern(c), back) orelse return null; dir = 0; size = .line; }, else => { a.i -= 1; return r; }, } } if (dir != 0) r = a.number(r, 1, dir, .line) orelse return null; return r; } fn pattern(a: *Addr, delim: u8) []const u8 { const s = a.i; while (a.i < a.expr.len) { const c = a.expr[a.i]; if (c == '\n') break; a.i += 1; if (c == '\\') { if (a.i < a.expr.len) a.i += 1; continue; } if (c == delim) return a.expr[s .. a.i - 1]; } return a.expr[s..a.i]; } fn number(a: *Addr, r_in: Range, n: u32, dir: u8, size: Size) ?Range { var r = r_in; if (size == .char) { var off: i64 = n; if (dir == '+') { off = @as(i64, r.q1) + n; } else if (dir == '-') { if (r.q0 == 0 and n > 0) r.q0 = clip(a.text.len); off = @as(i64, r.q0) - n; } if (off < 0 or off > @as(i64, @intCast(a.text.len))) { a.err = e_range; return null; } const g = clip(modal.graphemeStart(a.text, @intCast(off))); return .{ .q0 = g, .q1 = g }; } var line: i64 = n; var q0: usize = r.q0; var q1: usize = r.q1; switch (dir) { '-' => { if (q0 < a.text.len) while (q0 > 0 and a.text[q0 - 1] != '\n') { q0 -= 1; }; q1 = q0; while (line > 0 and q0 > 0) { if (a.text[q0 - 1] == '\n') { line -= 1; q1 = q0; } q0 -= 1; } if (line > 1) { a.err = e_range; return null; } while (q0 > 0 and a.text[q0 - 1] != '\n') q0 -= 1; return .{ .q0 = clip(q0), .q1 = clip(q1) }; }, '+' => { if (q1 > 0) while (q1 < a.text.len and a.text[q1 - 1] != '\n') { q1 += 1; }; q0 = q1; }, else => { q0 = 0; q1 = 0; }, } while (line > 0 and q1 < a.text.len) { const ch = a.text[q1]; q1 += 1; if (ch == '\n' or q1 == a.text.len) { line -= 1; if (line > 0) q0 = q1; } } if (line > 0) { a.err = e_range; return null; } return .{ .q0 = clip(q0), .q1 = clip(q1) }; } /// acme's regexp(): forward from the end of `r` to the limit, wrapping /// to the start of the text when there is none; backward, the last /// match ending by the start of `r`, else the last one in the text. fn regexp(a: *Addr, r: Range, pat: []const u8, back: bool) ?Range { if (pat.len == 0) { a.err = "no previous regular expression"; return null; } var rx = regexp_.Regex.compile(pat) catch { a.err = e_regexp; return null; }; const found = if (back) found: { var last: ?Range = null; var before: ?Range = null; var at: usize = 0; while (at <= a.text.len) { const m = (rx.find(a.text, at, a.text.len, a.text.len) catch { a.err = e_slow; return null; }) orelse break; const found_range: Range = .{ .q0 = clip(m.start), .q1 = clip(m.end) }; if (found_range.q1 <= r.q0) before = found_range; last = found_range; at = if (m.end > m.start) m.end else m.end + 1; } break :found before orelse last; } else found: { const hi = if (a.lim) |l| @min(@as(usize, l.q1), a.text.len) else a.text.len; const from = @min(@as(usize, r.q1), hi); const ahead = rx.find(a.text, from, hi, hi) catch { a.err = e_slow; return null; }; const m = ahead orelse (if (a.lim != null or from == 0) null else rx.find(a.text, 0, from - 1, hi) catch { a.err = e_slow; return null; }) orelse break :found null; break :found Range{ .q0 = clip(m.start), .q1 = clip(m.end) }; }; return found orelse { a.err = e_no_match; return null; }; } }; const testing = std.testing; const th = @import("testing.zig"); const tree = @import("tree.zig"); const Node = tree.Node; const E = tree.E; const Status = tree.Status; test "regular expressions search lines as sam's do, and a search wraps" { const p = try th.withFile(testing.allocator, "alpha beta\nbeta gamma\ngamma\n"); defer p.deinit(); const serial = th.serialOf(p); const addr = Node.of(serial, .addr); const Case = struct { from: []const u8, expr: []const u8, q0: u32, q1: u32 }; for ([_]Case{ // ^ and $ at every line's start and end, not only the text's .{ .from = "#0", .expr = "/^beta/", .q0 = 11, .q1 = 15 }, .{ .from = "#0", .expr = "/beta$/", .q0 = 6, .q1 = 10 }, .{ .from = "#0", .expr = "/^gamma$/", .q0 = 22, .q1 = 27 }, // . and [^...] stop at a newline; \n names one .{ .from = "#0", .expr = "/a.*/", .q0 = 0, .q1 = 10 }, .{ .from = "#0", .expr = "/a[^x]*/", .q0 = 0, .q1 = 10 }, .{ .from = "#0", .expr = "/a\\nbeta/", .q0 = 9, .q1 = 15 }, // leftmost; of the alternatives there, mvzr's first .{ .from = "#0", .expr = "/be|beta b/", .q0 = 6, .q1 = 8 }, .{ .from = "#0", .expr = "/gam|gamma/", .q0 = 16, .q1 = 19 }, .{ .from = "#0", .expr = "/(al)+pha?/", .q0 = 0, .q1 = 5 }, // `^` does not match where a search starts in the middle of a line .{ .from = "#1", .expr = "/^/", .q0 = 11, .q1 = 11 }, // a search past the last match wraps to the text's start .{ .from = "$", .expr = "/alpha/", .q0 = 0, .q1 = 5 }, // backward: the last match that ends by the start of dot .{ .from = "#16", .expr = "?beta?", .q0 = 11, .q1 = 15 }, .{ .from = "#0", .expr = "?gamma?", .q0 = 22, .q1 = 27 }, }) |c| { _ = th.wr(p, addr, c.from); const w = th.wr(p, addr, c.expr); try testing.expectEqual(Status.ok, w.reply.status); try testing.expectEqual(c.q0, p.panes[0].?.fs.addr.q0); try testing.expectEqual(c.q1, p.panes[0].?.fs.addr.q1); } // $ also matches at the end of text with no newline to end it const q = try th.withFile(testing.allocator, "last line"); defer q.deinit(); _ = th.wr(q, Node.of(th.serialOf(q), .addr), "#0"); try testing.expectEqual(Status.ok, th.wr(q, Node.of(th.serialOf(q), .addr), "/line$/").reply.status); try testing.expectEqual(@as(u32, 5), q.panes[0].?.fs.addr.q0); } test "a failed address leaves none, so data refuses rather than act at the last one" { const p = try th.withFile(testing.allocator, "one\ntwo\n"); defer p.deinit(); const serial = th.serialOf(p); const addr = Node.of(serial, .addr); const data = Node.of(serial, .data); _ = th.wr(p, addr, "#0,#3"); try testing.expectEqualStrings(e_no_match, th.wr(p, addr, "/zzz/").reply.ename); try testing.expectEqualStrings(pane_files.e_addr_failed, th.wr(p, data, "ONE").reply.ename); try testing.expectEqualStrings(pane_files.e_addr_failed, th.rd(p, data, 0, 64).reply.ename); try testing.expectEqualStrings(pane_files.e_addr_failed, th.rd(p, addr, 0, 64).reply.ename); try testing.expectEqual(E.INVAL, th.call(p, .{ .tag = 1, .op = .setattr, .node = Node.of(serial, .xdata), .truncate = true }).errno()); try testing.expectEqualStrings("one\ntwo\n", p.panes[0].?.file.?.content); // A good address, or a truncated addr, gives it one again. _ = th.wr(p, addr, "#0,#3"); try testing.expectEqual(Status.ok, th.wr(p, data, "ONE").reply.status); try testing.expectEqualStrings("ONE\ntwo\n", p.panes[0].?.file.?.content); _ = th.wr(p, addr, "99"); _ = th.call(p, .{ .tag = 1, .op = .setattr, .node = addr, .truncate = true }); try testing.expectEqual(Status.ok, th.wr(p, data, "0").reply.status); } test "the address language, form by form" { const gpa = testing.allocator; const p = try th.withFile(gpa, "one\ntwo\nthree\n"); defer p.deinit(); const serial = th.serialOf(p); const addr = Node.of(serial, .addr); const Case = struct { expr: []const u8, q0: u32, q1: u32 }; for ([_]Case{ .{ .expr = "#0", .q0 = 0, .q1 = 0 }, .{ .expr = "#5", .q0 = 5, .q1 = 5 }, .{ .expr = "0", .q0 = 0, .q1 = 0 }, .{ .expr = "1", .q0 = 0, .q1 = 4 }, .{ .expr = "2", .q0 = 4, .q1 = 8 }, .{ .expr = "$", .q0 = 14, .q1 = 14 }, .{ .expr = ",", .q0 = 0, .q1 = 14 }, .{ .expr = "1,2", .q0 = 0, .q1 = 8 }, .{ .expr = "#1,#4", .q0 = 1, .q1 = 4 }, .{ .expr = "2+1", .q0 = 8, .q1 = 14 }, .{ .expr = "$-1", .q0 = 8, .q1 = 14 }, .{ .expr = "/two/", .q0 = 4, .q1 = 7 }, .{ .expr = "/t.o/", .q0 = 4, .q1 = 7 }, .{ .expr = "1\n", .q0 = 0, .q1 = 4 }, }) |c| { _ = th.wr(p, addr, "#0"); const w = th.wr(p, addr, c.expr); try testing.expectEqual(Status.ok, w.reply.status); const got = th.rd(p, addr, 0, 64); var want: [32]u8 = undefined; try testing.expectEqualStrings( try std.fmt.bufPrint(&want, "{d:>11} {d:>11} ", .{ c.q0, c.q1 }), got.bytes, ); } _ = th.wr(p, addr, "1"); _ = th.wr(p, addr, "."); try testing.expectEqual(@as(u32, 0), p.panes[0].?.fs.addr.q0); try testing.expectEqual(@as(u32, 4), p.panes[0].?.fs.addr.q1); _ = th.wr(p, addr, "$"); _ = th.wr(p, addr, "?o?"); try testing.expectEqual(@as(u32, 6), p.panes[0].?.fs.addr.q0); // the `o` in "two" try testing.expectEqual(@as(u32, 7), p.panes[0].?.fs.addr.q1); // limit is its own file: copying addr onto it bounds the search, and // truncating it lifts the bound again. const limit = Node.of(serial, .limit); _ = th.wr(p, addr, "1"); try testing.expectEqualStrings("", th.rd(p, limit, 0, 64).bytes); var copied: [64]u8 = undefined; const pair = th.rd(p, addr, 0, 64).bytes; @memcpy(copied[0..pair.len], pair); _ = th.wr(p, limit, copied[0..pair.len]); try testing.expectEqual(@as(u32, 4), p.panes[0].?.fs.limit.?.q1); try testing.expectEqualStrings(" 0 4 ", th.rd(p, limit, 0, 64).bytes); _ = th.wr(p, addr, "#0"); try testing.expectEqual(E.INVAL, th.wr(p, addr, "/three/").errno()); _ = th.call(p, .{ .tag = 6, .op = .setattr, .node = limit, .truncate = true }); try testing.expect(p.panes[0].?.fs.limit == null); _ = th.wr(p, addr, "#0"); try testing.expectEqual(Status.ok, th.wr(p, addr, "/three/").reply.status); for ([_][]const u8{ "zzz", "#", "//", "/nomatch/", "1 2 3", "99", "/a\\" }) |bad| { _ = th.wr(p, addr, "#0"); try testing.expectEqual(E.INVAL, th.wr(p, addr, bad).errno()); } // Each failure says which it is. for ([_][2][]const u8{ .{ "zzz", e_syntax }, .{ "/nomatch/", e_no_match }, .{ "99", e_range }, .{ "#999", e_range }, .{ "/a[/", e_regexp }, .{ "/(a/", e_regexp }, .{ "/*a/", e_regexp }, }) |c| { _ = th.wr(p, addr, "#0"); try testing.expectEqualStrings(c[1], th.wr(p, addr, c[0]).reply.ename); } const nested = "," ** 4096; _ = th.wr(p, addr, "#0"); try testing.expectEqual(E.INVAL, th.wr(p, addr, nested).errno()); }