//! BASE 9P2000, ON THE WIRE AND NOTHING ELSE: the twenty-seven message types //! of the original protocol, encoded into a caller's buffer and decoded back //! out of one, with no allocator, no descriptor and no opinion about what any //! message means. //! //! WHY A SECOND CODEC in a tree that already has `src/detached/wire.zig`. That //! one is ours on both ends and can be renumbered by editing one file. This one //! is somebody else's: plan9port's `9p` command, Plan 9's own `mount`, Linux's //! v9fs and `ad` will all be talking to it, and not one of them will be //! rebuilt to suit us. So every number below is copied from a primary source //! with the file and line named, and the tests at the bottom assert LITERAL //! BYTES against `u9fs/convS2M.c` rather than only round-tripping — a codec //! that agrees with itself has proved nothing about interoperability. //! //! WHAT THIS DELIBERATELY IS NOT. There are three dialects; this is the first. //! * 9P2000.u adds a numeric errno to `Rerror`, `n_uid`/`extension` to //! `stat`, and Unix-flavoured `Tcreate`. We do not serve it. The tree is //! acme's and it is INVENTED — every error in it is one we chose the //! wording of, so a string is the whole error ABI and a number beside it //! would be a second spelling of a decision we already made //! (docs/registry.typ `9P-4`). It also costs a second dialect inside every //! one of the parsers below, because `.u` changes the LAYOUT of `Rerror` //! and `stat` rather than adding messages. //! * 9P2000.L replaces most of the protocol: `Tstatfs`, `Tlopen`, `Tgetattr`, //! `Tsymlink`, `Trename`, thirty-odd types and a POSIX file model. Our tree //! has no symlinks, no hard links, no device nodes and no block counts to //! report, so there is nothing on the other side of those messages to //! answer them with. //! * `Tsession`/`Tattach`-with-auth-blob from the 9P1 era, which u9fs still //! carries commented out (`convS2M.c:60-65,140-148`) and which no client //! built this century sends. //! //! TWO FACTS A READER MUST NOT GET WRONG, because both are silent when wrong: //! //! 1. `size[4]` INCLUDES ITSELF. `convS2M.c:216-224` computes `size` from //! `sizeS2M`, whose first line is `n += BIT32SZ; /* size */`, and then //! writes that number into the first four bytes. A reader that treats it //! as a payload length is four bytes out of step on every message and //! resynchronises never. //! //! 2. A `stat` HAS TWO LENGTHS IN FRONT OF IT. The record itself begins with //! `size[2]` which counts everything AFTER itself — `convD2M.c:48-51`, //! «note that length excludes count field itself», `PBIT16(p, ss-BIT16SZ)` //! — and `Rstat`/`Twstat` then wrap the whole record in ANOTHER `[2]` //! count, which is why Linux reads `Rstat` with the format string `"wS"` //! and throws the first `w` away into a variable literally named `ignored` //! (`linux/net/9p/client.c:1617,1633`), and writes `Twstat` as `"dwS"` //! (`client.c:1776`). So the outer count is `Stat.size() + 2`, never //! `Stat.size()`. `Stat` below owns both numbers and the test //! "9p: the stat double length" is the one that would catch it. //! //! FREESTANDING. No libc, no OS, no allocator, no threads: this file imports //! `std` for `mem.readInt`/`writeInt` and `debug.assert` and nothing more, so //! it compiles for `wasm32-freestanding` and for the board's //! `riscv32-freestanding` exactly as `wire.zig` does. Every integer on the wire //! has an explicit width and is little-endian; no `usize` reaches it, and no //! Zig struct is ever `@bitCast` onto it. Decoding BORROWS: every `[]const u8` //! in a decoded `Msg` points into the caller's buffer, which stays alive until //! the reply is written. //! //! MALFORMED INPUT IS REFUSED. This parser is fed by a socket, and a message //! misread rather than refused is an out-of-bounds index. Nothing below indexes //! without first checking; `Error` names every way a stream can be wrong. //! //! THE SERVER HALF IS APPENDED TO THIS FILE, the way `fuse.zig` keeps its wire //! structs and its transport together: a fid table, the dispatch onto //! `acmefs.zig`'s nine operations, and the msize handshake. Codec first, then //! the seam. Keeping them in one file is what makes it possible to change a //! layout and its only caller in one diff. //! //! Verified against `u9fs` (`fcall.h`, `convS2M.c`, `convM2S.c`, `convD2M.c`, //! `convM2D.c`), `linux/net/9p/{protocol,client}.c`, and //! `ad/crates/ninep/src/sansio/protocol.rs`. const std = @import("std"); const assert = std.debug.assert; pub const Error = error{ /// The message ended inside a field, or `size` claims more bytes than the /// caller handed over. Both are "not all of it has arrived", which is what /// a stream reader wants to hear: buffer more and ask again. Truncated, /// A count larger than this protocol admits: `nwname > MAXWELEM`, a string /// past 64 KiB, a `stat` past 64 KiB. Refused before anything is indexed. Overlong, /// A type byte 9P2000 does not define, or defines as illegal (`Terror`). BadTag, /// A field carrying a value it cannot mean — a `size` smaller than a /// header, which is a number no encoder can have produced. BadValue, /// The message was decoded and bytes were left over, either inside `size` /// or after it. A message that says more than its layout has room for is /// not this message. Trailing, /// The encoder ran out of caller-supplied buffer. Nothing was written. NoSpace, }; // --------------------------------------------------------------------------- // message types // --------------------------------------------------------------------------- /// The type byte. Numbers from `u9fs/fcall.h:74-105`, which is the definitive /// list: `Tversion = 100` and every name after it takes the next value, so the /// gap at 106 is load-bearing and the enum below spells it rather than skipping /// it silently. /// /// Non-exhaustive for the same reason `fuse.zig`'s `Opcode` is: `@enumFromInt` /// of an unlisted value into an exhaustive enum is undefined behaviour, which /// is the one bug in a protocol decoder that cannot be diagnosed from outside. /// A `.u` or `.L` client will hand us `Tstatfs = 8` or `Tlopen = 12`; that must /// arrive as a value we can refuse (`decode` returns `error.BadTag`) rather /// than as UB. /// /// TWENTY-SEVEN REAL TYPES: thirteen T/R pairs, plus `Rerror`, which is a reply /// with no request. `Terror = 106` is the twenty-eighth number and is defined /// as illegal by the protocol — a client cannot ask for an error — so it is /// listed to keep the numbering honest and refused by name in `decode`. pub const Type = enum(u8) { tversion = 100, rversion = 101, tauth = 102, rauth = 103, tattach = 104, rattach = 105, /// «Terror = 106, /* illegal */» — `fcall.h:82`. Never sent, never /// accepted; here so that nobody re-derives 107 for `Rerror` by counting. terror = 106, rerror = 107, tflush = 108, rflush = 109, twalk = 110, rwalk = 111, topen = 112, ropen = 113, tcreate = 114, rcreate = 115, tread = 116, rread = 117, twrite = 118, rwrite = 119, tclunk = 120, rclunk = 121, tremove = 122, rremove = 123, tstat = 124, rstat = 125, twstat = 126, rwstat = 127, _, }; /// T-messages are EVEN, R-messages are ODD, all the way from `Tversion = 100` /// to `Rwstat = 127` (`fcall.h:74-105`), because the enum assigns each T an /// even number and lets its R take the next. So one byte tells a reader which /// direction a message is travelling, which makes a stream carrying both /// SELF-DEMUXING: `drawterm`'s single descriptor has requests going one way and /// replies coming back on it, and the parity alone separates them. /// /// PaRDeS does not rely on that today — a connection has one role per side /// (docs/9p.typ §"Layering"), so a server only ever reads T and a client only /// ever reads R, and each refuses the other by name. This is here because the /// ENCODING GUARANTEES it and a future 9P-inside-the-wire arrangement over the /// board's UART would want it, and because a hand-typed number that breaks the /// parity is a bug the test at the bottom catches for free. pub fn isT(t: Type) bool { return @intFromEnum(t) % 2 == 0; } // --------------------------------------------------------------------------- // constants // --------------------------------------------------------------------------- /// `size[4] type[1] tag[2]`, and `size` counts these seven bytes too. Public /// because the server half sizes its reply payloads against it: the largest /// `Rread` that fits an msize is `msize - header_len - 4`. pub const header_len: usize = 4 + 1 + 2; /// `QIDSZ` — `fcall.h:64`, `BIT8SZ+BIT32SZ+BIT64SZ`. pub const qid_len: usize = 1 + 4 + 8; /// `STATFIXLEN` — `fcall.h:66-68`. The fixed part of a `stat` INCLUDING its own /// leading `size[2]` and the four string count prefixes, excluding the string /// bytes. `BIT16SZ + QIDSZ + 5*BIT16SZ + 4*BIT32SZ + BIT64SZ` = 49. pub const stat_fixed: usize = 2 + qid_len + 5 * 2 + 4 * 4 + 8; /// `NOTAG` — `fcall.h:71`. The tag on `Tversion`/`Rversion`, which is the one /// exchange that happens before tags mean anything. Note that `fcall.h` writes /// it `~0U` and the wire field is two bytes, so it is 0xFFFF and not 0xFFFFFFFF. pub const notag: u16 = 0xFFFF; /// `NOFID` — `fcall.h:121-123`. `Tattach.afid` when no authentication fid was /// established, which is our only use of it: we serve `Tauth` a refusal. pub const nofid: u32 = 0xFFFF_FFFF; /// `MAXWELEM` — `fcall.h:2`. The most path elements one `Twalk` may carry, and /// a hard protocol bound rather than a buffer size: `convS2M.c:279-280` and /// `convM2S.c:170-171` both return failure above it, so a 17-element walk is /// refused by every implementation and must be split by the client. pub const max_welem: usize = 16; /// The smallest msize we may agree to. NOT from the protocol — 9P has no floor, /// and Plan 9's devmnt, plan9port's `9p` and our own client all accept 512 /// (docs/registry.typ, `linux/net/9p/client.c:840-843`). This number exists /// because the LINUX KERNEL refuses to mount below it, and a mount that fails /// with `EINVAL` and no message is the worst diagnostic in the set. pub const min_msize: u32 = 4096; /// `IOHDRSZ` — `fcall.h:72`, «ample room for Twrite/Rread header (iounit)». /// The real `Rread` header is 11 bytes (`size[4] type[1] tag[2] count[4]`) and /// `Twrite`'s is 23; 24 is the slack both ends have agreed to reserve for /// thirty years, and `iounit` is quoted to clients as `msize - iohdrsz`. pub const iohdrsz: u32 = 24; /// `ERRMAX` — Plan 9's `libc.h:146`. The buffer a Plan 9 client has for an /// error string. ADVISORY here: `decode` does not refuse a longer `Rerror`, /// because refusing a peer's error message is the least useful moment to /// discover a length limit. The server half truncates its own to this. pub const errmax: usize = 128; // Qid type bits — `u9fs/plan9.h:156-161`, cross-checked against // `linux/include/net/9p/9p.h:344-352` which adds QTTMP = 0x04. These are the // top five bits of `Stat.mode` shifted down 24; see `dmdir` below. pub const qtdir: u8 = 0x80; pub const qtappend: u8 = 0x40; pub const qtexcl: u8 = 0x20; /// 0x10 is `QTMOUNT`, a mounted channel — a thing only a Plan 9 kernel has, and /// the reason the mode bits below have a gap at bit 28. pub const qtmount: u8 = 0x10; pub const qtauth: u8 = 0x08; pub const qttmp: u8 = 0x04; /// «plain file» — `plan9.h:161`. Zero, so a `Qid.type` of 0 is not "unset". pub const qtfile: u8 = 0x00; // Mode bits — `u9fs/plan9.h:164-170`, with DMAUTH and DMTMP from // `ad/crates/ninep/src/sansio/protocol.rs:486-495`: «bit 27 (DMAUTH) ... bit 26 // (DMTMP) ... (Bit 28 is skipped for historical reasons)». That skipped bit is // `DMMOUNT`, which is why the top five type bits are not the top five mode // bits: they are DMDIR, DMAPPEND, DMEXCL, (gap), DMAUTH, DMTMP reproduced from // the top down into `Qid.type` as QTDIR, QTAPPEND, QTEXCL, QTAUTH, QTTMP. pub const dmdir: u32 = 0x8000_0000; pub const dmappend: u32 = 0x4000_0000; pub const dmexcl: u32 = 0x2000_0000; pub const dmmount: u32 = 0x1000_0000; pub const dmauth: u32 = 0x0800_0000; pub const dmtmp: u32 = 0x0400_0000; /// The rwx triples, and the ONLY part of `mode` that is a Unix permission. A /// server that hands the high bits to `chmod`, or a client that hands the low /// nine to a type test, has confused the two halves of one word. pub const dmperm: u32 = 0o777; comptime { // The three widths every offset below is derived from. A drifted number // here is a codec that agrees with nothing, so make it a compile error. assert(header_len == 7); assert(qid_len == 13); assert(stat_fixed == 49); // Parity is the protocol's, not a convention we maintain by hand. for (std.enums.values(Type)) |t| { const even = @intFromEnum(t) % 2 == 0; assert(isT(t) == even); assert(std.mem.startsWith(u8, @tagName(t), if (even) "t" else "r")); } } // --------------------------------------------------------------------------- // qid // --------------------------------------------------------------------------- /// The server's name for a file: `type[1] version[4] path[8]`, thirteen bytes, /// `convS2M.c:18-29`. Two files are the same file if and only if their qids /// are equal, which is the whole contract — `path` identifies the file for the /// life of the connection and `version` changes on every write, so a client /// caches against the pair and never against a pathname. pub const Qid = struct { /// `qt*` bits. The high bits of `Stat.mode` shifted down 24. type: u8, version: u32, path: u64, /// Writes thirteen bytes and returns them. Takes the whole buffer and /// returns the used slice, so a caller can chain without arithmetic. pub fn encode(self: Qid, buf: []u8) Error![]u8 { if (buf.len < qid_len) return error.NoSpace; buf[0] = self.type; std.mem.writeInt(u32, buf[1..5], self.version, .little); std.mem.writeInt(u64, buf[5..13], self.path, .little); return buf[0..qid_len]; } /// Reads thirteen bytes. Refuses a shorter buffer rather than reading one: /// `gqid` in `convM2S.c:26-37` returns nil for exactly this case. pub fn decode(bytes: []const u8) Error!Qid { if (bytes.len < qid_len) return error.Truncated; return .{ .type = bytes[0], .version = std.mem.readInt(u32, bytes[1..5], .little), .path = std.mem.readInt(u64, bytes[5..13], .little), }; } }; // --------------------------------------------------------------------------- // stat // --------------------------------------------------------------------------- /// One directory entry, and the payload of `Rstat` and `Twstat`. Layout from /// `convD2M.c:56-83`: /// /// ``` /// size[2] type[2] dev[4] qid[13] mode[4] atime[4] mtime[4] length[8] /// name[s] uid[s] gid[s] muid[s] /// ``` /// /// where `[s]` is `n[2]` plus n bytes of UTF-8, NOT NUL-terminated. `size` /// counts everything after itself, so the record occupies `size() + 2` bytes; /// see the module header for why that matters twice over. /// /// `type` and `dev` are Plan 9 kernel device identifiers and are meaningless /// off Plan 9 — u9fs sends zeros and so do we, but they are on the wire because /// the layout is fixed. `muid` is the uid of the last modifier; for a synthetic /// tree it is whoever attached. /// /// A `Twstat` uses the sentinel "don't touch" values that acme(4) and /// `stat(5)` specify: an empty string, an all-ones integer. Nothing here /// interprets them; that is the server half's job. pub const Stat = struct { type: u16, dev: u32, qid: Qid, mode: u32, atime: u32, mtime: u32, length: u64, name: []const u8, uid: []const u8, gid: []const u8, muid: []const u8, /// The value that goes in the leading `size[2]`: every byte of the record /// EXCEPT those two. `convD2M.c:46-50` computes `ss = STATFIXLEN + ns` and /// then writes `ss - BIT16SZ`, so this is `stat_fixed - 2` plus the four /// string bodies. Written out field by field rather than as 47, because a /// number nobody can check against a layout is a comment that rots. /// /// Fallible: the prefix is two bytes, so a record whose strings do not fit /// a u16 has no legal encoding and must be refused rather than wrapped. pub fn size(self: Stat) Error!u16 { const n = 2 + // type 4 + // dev qid_len + // qid: type[1] version[4] path[8] 4 + // mode 4 + // atime 4 + // mtime 8 + // length 2 + self.name.len + 2 + self.uid.len + 2 + self.gid.len + 2 + self.muid.len; assert(n >= stat_fixed - 2); if (n > std.math.maxInt(u16)) return error.Overlong; return @intCast(n); } /// Writes `size[2]` and the record, and returns the `size() + 2` bytes of /// it. `assert` at the end is `convD2M.c:85-86`'s `if(ss != p - buf)`: the /// two arithmetics are written separately and must agree. pub fn encode(self: Stat, buf: []u8) Error![]u8 { const n = try self.size(); const total = @as(usize, n) + 2; if (buf.len < total) return error.NoSpace; var w: Writer = .init(buf[0..total]); try w.putU16(n); try w.putU16(self.type); try w.putU32(self.dev); try w.putQid(self.qid); try w.putU32(self.mode); try w.putU32(self.atime); try w.putU32(self.mtime); try w.putU64(self.length); try w.putString(self.name); try w.putString(self.uid); try w.putString(self.gid); try w.putString(self.muid); assert(w.n == total); return buf[0..total]; } /// Decodes exactly one record from `bytes`, which must be the whole of it — /// prefix included — and nothing more. The strings BORROW from `bytes`. /// /// The equality check on the prefix is the second half of `statcheck` /// (`convM2D.c:14-22`: walk the four counts, then `if(buf != ebuf) return /// -1`). It is what makes the double length safe: the caller has already /// bounded `bytes` by the OUTER count, so demanding that the INNER count /// agree refuses the classic off-by-two in both directions instead of /// trusting whichever one the sender got right. pub fn decode(bytes: []const u8) Error!Stat { var r: Reader = .init(bytes); const n = try r.getU16(); const body = bytes.len - 2; if (n > body) return error.Truncated; if (n < body) return error.Trailing; const self: Stat = .{ .type = try r.getU16(), .dev = try r.getU32(), .qid = try r.getQid(), .mode = try r.getU32(), .atime = try r.getU32(), .mtime = try r.getU32(), .length = try r.getU64(), .name = try r.getString(), .uid = try r.getString(), .gid = try r.getString(), .muid = try r.getString(), }; try r.end(); return self; } }; // --------------------------------------------------------------------------- // messages // --------------------------------------------------------------------------- /// Every message base 9P2000 defines, with the fields it actually carries. /// Layouts from `convS2M.c:231-419` and `convM2S.c:73-375`, which are the two /// halves of the same table and disagree nowhere. /// /// The tag names are the type names, so `msgType` is a mechanical mapping and /// not a table somebody maintains; a variant added here without a `Type` is a /// compile error. /// /// NOT IN HERE: the message tag. A `Msg` is a message's CONTENT, and the tag is /// the transport's matching of a reply to a request — it is a parameter of /// `encode` and a field of `Decoded`. Putting it in the union would mean every /// server handler that builds a reply has to remember to copy it. /// /// `Twalk` is the large variant at sixteen slices, so `Msg` is around 280 bytes /// on a 64-bit host. That is a value passed by const pointer in practice and it /// buys the thing that matters: a walk decodes with no allocator and no bound /// the caller has to have guessed. pub const Msg = union(enum) { /// The first exchange, tagged `notag`. `msize` is the largest message /// either end will send, INCLUDING the seven-byte header; `version` is /// "9P2000" or a string starting with it. tversion: struct { msize: u32, version: []const u8 }, /// The server's answer: `msize` no larger than the client's, and `version` /// either "9P2000" or the literal "unknown" — which is a successful reply /// meaning "no dialect in common", not an `Rerror`. rversion: struct { msize: u32, version: []const u8 }, tauth: struct { afid: u32, uname: []const u8, aname: []const u8 }, /// `aqid` and not `qid`: `fcall.h:44` gives `Rauth` its own field, and /// `convS2M.c:368-370` writes it. Same thirteen bytes, different meaning — /// the qid of the auth FILE, not of the tree. rauth: struct { aqid: Qid }, /// `afid` is `nofid` when the client did not authenticate. tattach: struct { fid: u32, afid: u32, uname: []const u8, aname: []const u8 }, rattach: struct { qid: Qid }, /// A STRING and nothing else. Base 9P2000 has no numeric error code; the /// `errno` field is 9P2000.u's, which this file does not serve. See the /// module header. rerror: struct { ename: []const u8 }, /// `oldtag` is a u16 like every tag, even though `fcall.h:14` declares /// `oldtag` as u32 — `convS2M.c:251-253` writes it with `PBIT16`. tflush: struct { oldtag: u16 }, rflush: void, /// `wname[0..nwname]` are the path elements; anything past `nwname` is /// undefined and neither encoded nor compared. A zero-element walk is /// legal and means "clone `fid` into `newfid`". twalk: struct { fid: u32, newfid: u32, nwname: u16, wname: [max_welem][]const u8 = @splat(""), }, /// `nwqid` may be SHORTER than the request's `nwname`: a partial walk is a /// successful `Rwalk` with fewer qids, and only a failure on the FIRST /// element is an `Rerror`. rwalk: struct { nwqid: u16, wqid: [max_welem]Qid = @splat(.{ .type = 0, .version = 0, .path = 0 }), }, /// `mode` is OREAD/OWRITE/ORDWR/OEXEC plus OTRUNC/ORCLOSE, one byte. topen: struct { fid: u32, mode: u8 }, /// `iounit`: the largest atomic read or write, or 0 for "no promise". We /// quote `msize - iohdrsz`. ropen: struct { qid: Qid, iounit: u32 }, /// `perm` is the full mode word — `dmdir` and friends in the high bits, /// `dmperm` in the low nine. tcreate: struct { fid: u32, name: []const u8, perm: u32, mode: u8 }, rcreate: struct { qid: Qid, iounit: u32 }, tread: struct { fid: u32, offset: u64, count: u32 }, /// `count[4]` then the bytes, held as one slice because the count is the /// slice's length and two ways to say one number is one way to disagree. /// A reply longer than the request's `count` is a hard `-EIO` to Linux /// (`net/9p/client.c:1475-1479`), so the server half clamps and this codec /// carries whatever it is given. rread: struct { data: []const u8 }, twrite: struct { fid: u32, offset: u64, data: []const u8 }, /// The count actually written, which may be short. rwrite: struct { count: u32 }, tclunk: struct { fid: u32 }, rclunk: void, tremove: struct { fid: u32 }, rremove: void, tstat: struct { fid: u32 }, /// Carries a decoded `Stat`, not a blob, so the double length is computed /// in one place — `encode` derives the outer count from `stat.size() + 2` /// and `decode` demands they agree. u9fs keeps `nstat` and a `uchar*` here /// (`fcall.h:40-41`) and pays for it with `statcheck` as a separate call /// every caller must remember. rstat: struct { stat: Stat }, twstat: struct { fid: u32, stat: Stat }, rwstat: void, /// The type byte this message travels as. Mechanical, by name, so a /// variant cannot acquire the wrong number. pub fn msgType(msg: Msg) Type { return switch (msg) { inline else => |_, t| @field(Type, @tagName(t)), }; } }; /// One complete message off the wire: its tag and its content. The tag is /// separate for the reason `Msg`'s doc gives — a reply reuses the request's tag /// and never looks inside it. pub const Decoded = struct { tag: u16, msg: Msg, }; /// How many bytes this message will be, once the caller has enough of it. /// `null` when there are fewer than four, which is the only answer a stream /// reader can act on: read more. /// /// Deliberately UNVALIDATED. It is the raw `size` field, and it is peeked /// before the type byte has necessarily arrived, so there is nothing here to /// check it against. `decode` does the refusing; this only says how much to /// buffer, and a caller that compares the answer to its negotiated msize /// refuses an absurd claim before growing anything. pub fn frameLen(prefix: []const u8) ?u32 { if (prefix.len < 4) return null; return std.mem.readInt(u32, prefix[0..4], .little); } /// The whole message's byte count, `size` included, which IS the value of the /// `size` field. `sizeS2M` in `convS2M.c:38-208`, in the same order, so the two /// can be read side by side. /// /// Computed before a single byte is written, which is what makes `encode`'s /// `NoSpace` clean: a caller whose buffer is one byte short gets an error and /// an untouched buffer, not a half-written message. fn totalLen(msg: Msg) Error!usize { const body: usize = switch (msg) { .tversion => |m| 4 + try stringLen(m.version), .rversion => |m| 4 + try stringLen(m.version), .tauth => |m| 4 + try stringLen(m.uname) + try stringLen(m.aname), .rauth => qid_len, .tattach => |m| 4 + 4 + try stringLen(m.uname) + try stringLen(m.aname), .rattach => qid_len, .rerror => |m| try stringLen(m.ename), .tflush => 2, .rflush => 0, .twalk => |m| blk: { // The bound is the protocol's, and both halves of u9fs return // failure above it (`convS2M.c:279`, `convM2S.c:170`). On this side // it is a caller bug — the array is sixteen long — so it asserts. assert(m.nwname <= max_welem); var n: usize = 4 + 4 + 2; for (m.wname[0..m.nwname]) |name| n += try stringLen(name); break :blk n; }, .rwalk => |m| blk: { assert(m.nwqid <= max_welem); break :blk 2 + @as(usize, m.nwqid) * qid_len; }, .topen => 4 + 1, .ropen => qid_len + 4, .tcreate => |m| 4 + try stringLen(m.name) + 4 + 1, .rcreate => qid_len + 4, .tread => 4 + 8 + 4, .rread => |m| try dataLen(m.data), .twrite => |m| 4 + 8 + try dataLen(m.data), .rwrite => 4, .tclunk => 4, .rclunk => 0, .tremove => 4, .rremove => 0, .tstat => 4, // THE DOUBLE LENGTH, in the one place it is computed: the outer count, // then the record, whose own prefix is inside `size() + 2`. .rstat => |m| 2 + 2 + @as(usize, try m.stat.size()), .twstat => |m| 4 + 2 + 2 + @as(usize, try m.stat.size()), .rwstat => 0, }; const total = header_len + body; if (total > std.math.maxInt(u32)) return error.Overlong; return total; } /// `stringsz` — `convS2M.c:31-36`. The count is two bytes, so a longer string /// has no encoding and is refused here rather than truncated silently. fn stringLen(s: []const u8) Error!usize { if (s.len > std.math.maxInt(u16)) return error.Overlong; return 2 + s.len; } /// The same for a `count[4]` payload: `Rread`'s and `Twrite`'s data. fn dataLen(d: []const u8) Error!usize { if (d.len > std.math.maxInt(u32)) return error.Overlong; return 4 + d.len; } /// Encodes one message into `buf` and returns the bytes of it, which start at /// `buf[0]` and are exactly `size` long. /// /// `tag` is a parameter and not a field of `Msg`: a server handler builds a /// reply and the transport supplies the request's tag, so the two cannot drift. /// `notag` on anything but `Tversion`/`Rversion` is the caller's business. pub fn encode(msg: Msg, tag: u16, buf: []u8) Error![]u8 { const total = try totalLen(msg); if (total > buf.len) return error.NoSpace; // The writer is bounded to `total` and not to `buf`, so a disagreement // between `totalLen` and the field walk below cannot scribble past the // message — it becomes `NoSpace` here or the assert at the end. var w: Writer = .init(buf[0..total]); try w.putU32(@intCast(total)); try w.putByte(@intFromEnum(msg.msgType())); try w.putU16(tag); switch (msg) { .tversion => |m| { try w.putU32(m.msize); try w.putString(m.version); }, .rversion => |m| { try w.putU32(m.msize); try w.putString(m.version); }, .tauth => |m| { try w.putU32(m.afid); try w.putString(m.uname); try w.putString(m.aname); }, .rauth => |m| try w.putQid(m.aqid), .tattach => |m| { try w.putU32(m.fid); try w.putU32(m.afid); try w.putString(m.uname); try w.putString(m.aname); }, .rattach => |m| try w.putQid(m.qid), .rerror => |m| try w.putString(m.ename), .tflush => |m| try w.putU16(m.oldtag), .rflush => {}, .twalk => |m| { try w.putU32(m.fid); try w.putU32(m.newfid); try w.putU16(m.nwname); for (m.wname[0..m.nwname]) |name| try w.putString(name); }, .rwalk => |m| { try w.putU16(m.nwqid); for (m.wqid[0..m.nwqid]) |qid| try w.putQid(qid); }, .topen => |m| { try w.putU32(m.fid); try w.putByte(m.mode); }, .ropen => |m| { try w.putQid(m.qid); try w.putU32(m.iounit); }, .tcreate => |m| { try w.putU32(m.fid); try w.putString(m.name); try w.putU32(m.perm); try w.putByte(m.mode); }, .rcreate => |m| { try w.putQid(m.qid); try w.putU32(m.iounit); }, .tread => |m| { try w.putU32(m.fid); try w.putU64(m.offset); try w.putU32(m.count); }, .rread => |m| { try w.putU32(@intCast(m.data.len)); try w.putBytes(m.data); }, .twrite => |m| { try w.putU32(m.fid); try w.putU64(m.offset); try w.putU32(@intCast(m.data.len)); try w.putBytes(m.data); }, .rwrite => |m| try w.putU32(m.count), .tclunk => |m| try w.putU32(m.fid), .rclunk => {}, .tremove => |m| try w.putU32(m.fid), .rremove => {}, .tstat => |m| try w.putU32(m.fid), .rstat => |m| { // Outer count first: the whole record, its own prefix included. try w.putU16(try m.stat.size() + 2); try w.putStat(m.stat); }, .twstat => |m| { try w.putU32(m.fid); try w.putU16(try m.stat.size() + 2); try w.putStat(m.stat); }, .rwstat => {}, } // `convS2M.c:420-421`: `if(size != p-ap) return 0`. The two arithmetics are // deliberately separate and this is the only thing that keeps them honest. assert(w.n == total); return buf[0..total]; } /// Decodes exactly one complete message. `bytes` must be the message and /// nothing else — `frameLen` is how a reader knows where that ends — and every /// slice in the result BORROWS from it. /// /// Four ways this refuses, in the order the checks run, because the order is /// what makes a stream reader's life simple: /// * fewer than seven bytes, or `size` past the end -> `Truncated`, meaning /// "come back with more". /// * `size` short of the end -> `Trailing`. Two messages were handed over as /// one, which is a framing bug in the caller, not a short read. /// * a `size` that cannot hold a header -> `BadValue`. No encoder produced it. /// * a type byte 9P2000 does not define, or defines illegal -> `BadTag`. pub fn decode(bytes: []const u8) Error!Decoded { if (bytes.len < header_len) return error.Truncated; const size = std.mem.readInt(u32, bytes[0..4], .little); // `convM2S.c:65-66` refuses this too, and it must be refused BEFORE the // comparison against `bytes.len`: a size of 3 on a 3-byte buffer would // otherwise slice a header out of nothing. if (size < header_len) return error.BadValue; if (size > bytes.len) return error.Truncated; if (size < bytes.len) return error.Trailing; const t: Type = @enumFromInt(bytes[4]); const tag = std.mem.readInt(u16, bytes[5..7], .little); // Bounded by `size` and not by `bytes`, which is the same thing here only // because of the two checks above; keep it explicit so it stays true if a // caller is ever allowed to pass a longer buffer. var r: Reader = .init(bytes[header_len..size]); const msg: Msg = switch (t) { .tversion => .{ .tversion = .{ .msize = try r.getU32(), .version = try r.getString() } }, .rversion => .{ .rversion = .{ .msize = try r.getU32(), .version = try r.getString() } }, .tauth => .{ .tauth = .{ .afid = try r.getU32(), .uname = try r.getString(), .aname = try r.getString(), } }, .rauth => .{ .rauth = .{ .aqid = try r.getQid() } }, .tattach => .{ .tattach = .{ .fid = try r.getU32(), .afid = try r.getU32(), .uname = try r.getString(), .aname = try r.getString(), } }, .rattach => .{ .rattach = .{ .qid = try r.getQid() } }, .rerror => .{ .rerror = .{ .ename = try r.getString() } }, .tflush => .{ .tflush = .{ .oldtag = try r.getU16() } }, .rflush => .rflush, .twalk => blk: { var m: Msg = .{ .twalk = .{ .fid = try r.getU32(), .newfid = try r.getU32(), .nwname = try r.getU16(), } }; // Checked before the loop, so a hostile 65535 never reaches the // array. `convM2S.c:170-171` does the same and for the same reason. if (m.twalk.nwname > max_welem) return error.Overlong; for (m.twalk.wname[0..m.twalk.nwname]) |*name| name.* = try r.getString(); break :blk m; }, .rwalk => blk: { var m: Msg = .{ .rwalk = .{ .nwqid = try r.getU16() } }; if (m.rwalk.nwqid > max_welem) return error.Overlong; for (m.rwalk.wqid[0..m.rwalk.nwqid]) |*qid| qid.* = try r.getQid(); break :blk m; }, .topen => .{ .topen = .{ .fid = try r.getU32(), .mode = try r.getByte() } }, .ropen => .{ .ropen = .{ .qid = try r.getQid(), .iounit = try r.getU32() } }, .tcreate => .{ .tcreate = .{ .fid = try r.getU32(), .name = try r.getString(), .perm = try r.getU32(), .mode = try r.getByte(), } }, .rcreate => .{ .rcreate = .{ .qid = try r.getQid(), .iounit = try r.getU32() } }, .tread => .{ .tread = .{ .fid = try r.getU32(), .offset = try r.getU64(), .count = try r.getU32(), } }, .rread => .{ .rread = .{ .data = try r.getData() } }, .twrite => .{ .twrite = .{ .fid = try r.getU32(), .offset = try r.getU64(), .data = try r.getData(), } }, .rwrite => .{ .rwrite = .{ .count = try r.getU32() } }, .tclunk => .{ .tclunk = .{ .fid = try r.getU32() } }, .rclunk => .rclunk, .tremove => .{ .tremove = .{ .fid = try r.getU32() } }, .rremove => .rremove, .tstat => .{ .tstat = .{ .fid = try r.getU32() } }, .rstat => .{ .rstat = .{ .stat = try Stat.decode(try r.getBlob16()) } }, .twstat => .{ .twstat = .{ .fid = try r.getU32(), .stat = try Stat.decode(try r.getBlob16()), } }, .rwstat => .rwstat, // «Terror = 106, /* illegal */». A peer that sent one is not speaking // 9P2000, and every other unlisted byte is a `.u`/`.L` message or // noise. Both are refused here, which is also why `Type` is // non-exhaustive: this switch is reachable with any byte. .terror, _ => return error.BadTag, }; try r.end(); return .{ .tag = tag, .msg = msg }; } // --------------------------------------------------------------------------- // primitives // --------------------------------------------------------------------------- // // Explicit widths, little-endian, one field at a time. `GBIT*`/`PBIT*` in // `fcall.h:48-58` are the reference, and they are byte-at-a-time shifts for // exactly the reason this file does not blit a struct: the sender's word order // and padding are not the protocol. const Writer = struct { buf: []u8, n: usize = 0, fn init(buf: []u8) Writer { return .{ .buf = buf }; } /// `n <= buf.len` is the invariant every putter preserves, which is what /// makes the subtraction safe. fn room(w: *Writer, k: usize) Error![]u8 { if (w.buf.len - w.n < k) return error.NoSpace; defer w.n += k; return w.buf[w.n..][0..k]; } fn putByte(w: *Writer, v: u8) Error!void { (try w.room(1))[0] = v; } fn putU16(w: *Writer, v: u16) Error!void { std.mem.writeInt(u16, (try w.room(2))[0..2], v, .little); } fn putU32(w: *Writer, v: u32) Error!void { std.mem.writeInt(u32, (try w.room(4))[0..4], v, .little); } fn putU64(w: *Writer, v: u64) Error!void { std.mem.writeInt(u64, (try w.room(8))[0..8], v, .little); } fn putBytes(w: *Writer, v: []const u8) Error!void { @memcpy(try w.room(v.len), v); } /// `n[2]` then the bytes, NOT NUL-terminated — `pstring`, `convS2M.c:4-16`. /// The length was already refused by `stringLen` before anything was /// written, so this asserts rather than erroring: reaching it with a longer /// string means `totalLen` and this switch disagree. fn putString(w: *Writer, v: []const u8) Error!void { assert(v.len <= std.math.maxInt(u16)); try w.putU16(@intCast(v.len)); try w.putBytes(v); } fn putQid(w: *Writer, v: Qid) Error!void { _ = try v.encode(try w.room(qid_len)); } fn putStat(w: *Writer, v: Stat) Error!void { const total = @as(usize, try v.size()) + 2; _ = try v.encode(try w.room(total)); } }; const Reader = struct { bytes: []const u8, i: usize = 0, fn init(bytes: []const u8) Reader { return .{ .bytes = bytes }; } /// The one place this file indexes, and the one place it can refuse to. /// `i <= bytes.len` always, so the subtraction cannot wrap. fn take(r: *Reader, n: usize) Error![]const u8 { if (r.bytes.len - r.i < n) return error.Truncated; defer r.i += n; return r.bytes[r.i..][0..n]; } fn getByte(r: *Reader) Error!u8 { return (try r.take(1))[0]; } fn getU16(r: *Reader) Error!u16 { return std.mem.readInt(u16, (try r.take(2))[0..2], .little); } fn getU32(r: *Reader) Error!u32 { return std.mem.readInt(u32, (try r.take(4))[0..4], .little); } fn getU64(r: *Reader) Error!u64 { return std.mem.readInt(u64, (try r.take(8))[0..8], .little); } /// `gstring`, `convM2S.c:4-22`, minus the memmove: u9fs shuffles the bytes /// down over the count to make room for a '\0' because its callers are C /// string functions. Ours borrow, so the slice IS the string and the buffer /// is untouched. fn getString(r: *Reader) Error![]const u8 { return r.take(try r.getU16()); } /// `count[4]` then the bytes: `Rread`'s and `Twrite`'s payload. A count /// past the message is `Truncated` and not a clamp — Linux clamps here /// (`protocol.c:386-388`) and then has to catch the lie again in /// `client.c:1475-1479`. Refusing once is cheaper and says more. fn getData(r: *Reader) Error![]const u8 { return r.take(try r.getU32()); } /// `count[2]` then the bytes: the OUTER count of an `Rstat`/`Twstat` stat. /// Slicing exactly here is what lets `Stat.decode` insist that the record's /// own prefix agrees, which is the whole defence against the double length. fn getBlob16(r: *Reader) Error![]const u8 { return r.take(try r.getU16()); } fn getQid(r: *Reader) Error!Qid { return Qid.decode(try r.take(qid_len)); } fn end(r: *Reader) Error!void { if (r.i != r.bytes.len) return error.Trailing; } }; // --------------------------------------------------------------------------- // tests // --------------------------------------------------------------------------- // // Three obligations, and the third is the one that is usually skipped. // // 1. every message round-trips to an equal value, because a hand-written // codec is a codec whose two halves drift; // 2. every malformed shape is REFUSED and none of them panics, because this // parser is fed by a socket; // 3. the bytes are the RIGHT bytes. A round-trip test proves the encoder and // the decoder agree with each other and nothing about whether they agree // with plan9port's `9p`, which is who will actually be on the far end. So // four messages are hand-verified against `u9fs/convS2M.c` as literal // arrays with the line numbers attached. const testing = std.testing; fn roundTrip(buf: []u8, tag: u16, msg: Msg) !Msg { const bytes = try encode(msg, tag, buf); // The framing has to agree with the encoder before anything else is worth // checking: `size` includes itself, so this is also the regression test for // the first of the module header's two facts. try testing.expectEqual(bytes.len, frameLen(bytes).?); const got = try decode(bytes); try testing.expectEqual(tag, got.tag); try testing.expectEqual(msg.msgType(), got.msg.msgType()); try expectMsgEqual(msg, got.msg); return got.msg; } fn expectStatEqual(want: Stat, have: Stat) !void { try testing.expectEqual(want.type, have.type); try testing.expectEqual(want.dev, have.dev); try testing.expectEqual(want.qid, have.qid); try testing.expectEqual(want.mode, have.mode); try testing.expectEqual(want.atime, have.atime); try testing.expectEqual(want.mtime, have.mtime); try testing.expectEqual(want.length, have.length); try testing.expectEqualStrings(want.name, have.name); try testing.expectEqualStrings(want.uid, have.uid); try testing.expectEqualStrings(want.gid, have.gid); try testing.expectEqualStrings(want.muid, have.muid); } /// Field by field, because `std.meta.eql` is wrong here twice: a decoded slice /// points into the wire buffer and never compares equal by pointer, and /// `Twalk.wname` past `nwname` is scratch the decoder does not invent. fn expectMsgEqual(want: Msg, have: Msg) !void { switch (want) { .tversion => |w| { try testing.expectEqual(w.msize, have.tversion.msize); try testing.expectEqualStrings(w.version, have.tversion.version); }, .rversion => |w| { try testing.expectEqual(w.msize, have.rversion.msize); try testing.expectEqualStrings(w.version, have.rversion.version); }, .tauth => |w| { try testing.expectEqual(w.afid, have.tauth.afid); try testing.expectEqualStrings(w.uname, have.tauth.uname); try testing.expectEqualStrings(w.aname, have.tauth.aname); }, .rauth => |w| try testing.expectEqual(w.aqid, have.rauth.aqid), .tattach => |w| { try testing.expectEqual(w.fid, have.tattach.fid); try testing.expectEqual(w.afid, have.tattach.afid); try testing.expectEqualStrings(w.uname, have.tattach.uname); try testing.expectEqualStrings(w.aname, have.tattach.aname); }, .rattach => |w| try testing.expectEqual(w.qid, have.rattach.qid), .rerror => |w| try testing.expectEqualStrings(w.ename, have.rerror.ename), .tflush => |w| try testing.expectEqual(w.oldtag, have.tflush.oldtag), .rflush, .rclunk, .rremove, .rwstat => {}, .twalk => |w| { try testing.expectEqual(w.fid, have.twalk.fid); try testing.expectEqual(w.newfid, have.twalk.newfid); try testing.expectEqual(w.nwname, have.twalk.nwname); for (w.wname[0..w.nwname], have.twalk.wname[0..w.nwname]) |a, b| try testing.expectEqualStrings(a, b); }, .rwalk => |w| { try testing.expectEqual(w.nwqid, have.rwalk.nwqid); for (w.wqid[0..w.nwqid], have.rwalk.wqid[0..w.nwqid]) |a, b| try testing.expectEqual(a, b); }, .topen => |w| { try testing.expectEqual(w.fid, have.topen.fid); try testing.expectEqual(w.mode, have.topen.mode); }, .ropen => |w| { try testing.expectEqual(w.qid, have.ropen.qid); try testing.expectEqual(w.iounit, have.ropen.iounit); }, .tcreate => |w| { try testing.expectEqual(w.fid, have.tcreate.fid); try testing.expectEqualStrings(w.name, have.tcreate.name); try testing.expectEqual(w.perm, have.tcreate.perm); try testing.expectEqual(w.mode, have.tcreate.mode); }, .rcreate => |w| { try testing.expectEqual(w.qid, have.rcreate.qid); try testing.expectEqual(w.iounit, have.rcreate.iounit); }, .tread => |w| { try testing.expectEqual(w.fid, have.tread.fid); try testing.expectEqual(w.offset, have.tread.offset); try testing.expectEqual(w.count, have.tread.count); }, .rread => |w| try testing.expectEqualStrings(w.data, have.rread.data), .twrite => |w| { try testing.expectEqual(w.fid, have.twrite.fid); try testing.expectEqual(w.offset, have.twrite.offset); try testing.expectEqualStrings(w.data, have.twrite.data); }, .rwrite => |w| try testing.expectEqual(w.count, have.rwrite.count), .tclunk => |w| try testing.expectEqual(w.fid, have.tclunk.fid), .tremove => |w| try testing.expectEqual(w.fid, have.tremove.fid), .tstat => |w| try testing.expectEqual(w.fid, have.tstat.fid), .rstat => |w| try expectStatEqual(w.stat, have.rstat.stat), .twstat => |w| { try testing.expectEqual(w.fid, have.twstat.fid); try expectStatEqual(w.stat, have.twstat.stat); }, } } const sample_qid: Qid = .{ .type = qtdir, .version = 3, .path = 0x0102_0304_0506_0708 }; const sample_stat: Stat = .{ .type = 0, .dev = 0, .qid = sample_qid, .mode = dmdir | 0o755, .atime = 1, .mtime = 2, .length = 0, .name = "body", .uid = "goblin", .gid = "goblin", .muid = "goblin", }; test "9p: the type numbers and their parity are the protocol's own" { // Copied from `u9fs/fcall.h:74-105`. Asserted as literals because a // renumbering here is a codec that talks to nothing, and it must be a diff // somebody reads rather than a silent change. try testing.expectEqual(@as(u8, 100), @intFromEnum(Type.tversion)); try testing.expectEqual(@as(u8, 106), @intFromEnum(Type.terror)); try testing.expectEqual(@as(u8, 107), @intFromEnum(Type.rerror)); try testing.expectEqual(@as(u8, 126), @intFromEnum(Type.twstat)); try testing.expectEqual(@as(u8, 127), @intFromEnum(Type.rwstat)); // Twenty-eight numbers, 100..127 inclusive, no gaps and no strays. try testing.expectEqual(@as(usize, 28), std.enums.values(Type).len); for (std.enums.values(Type), 100..) |t, want| try testing.expectEqual(@as(u8, @intCast(want)), @intFromEnum(t)); try testing.expect(isT(.tversion)); try testing.expect(!isT(.rversion)); try testing.expect(isT(.twstat)); try testing.expect(!isT(.rwstat)); // `notag` is two bytes wide even though `fcall.h:71` writes `~0U`. try testing.expectEqual(@as(u16, 0xFFFF), notag); try testing.expectEqual(@as(u32, 0xFFFF_FFFF), nofid); try testing.expectEqual(@as(usize, 16), max_welem); } test "9p: a qid is thirteen bytes" { var buf: [32]u8 = undefined; const bytes = try sample_qid.encode(&buf); try testing.expectEqual(qid_len, bytes.len); try testing.expectEqual(@as(usize, 13), bytes.len); try testing.expectEqual(sample_qid, try Qid.decode(bytes)); // Twelve bytes is not a qid, and a decoder that read one anyway would be // reading the next field's first byte as the top of `path`. try testing.expectError(error.Truncated, Qid.decode(bytes[0..12])); try testing.expectError(error.NoSpace, sample_qid.encode(buf[0..12])); } test "9p: an encoded stat is size() + 2 bytes" { var buf: [256]u8 = undefined; const bytes = try sample_stat.encode(&buf); const n = try sample_stat.size(); try testing.expectEqual(@as(usize, n) + 2, bytes.len); // `STATFIXLEN - BIT16SZ` plus the four string bodies: 47 + 4 + 6 + 6 + 6. try testing.expectEqual(@as(u16, 69), n); try testing.expectEqual(stat_fixed - 2 + 22, n); // The prefix on the wire is the count EXCLUDING itself — `convD2M.c:48-51`. try testing.expectEqual(n, std.mem.readInt(u16, bytes[0..2], .little)); try expectStatEqual(sample_stat, try Stat.decode(bytes)); // Empty strings still cost their counts: 47 and nothing more. const bare: Stat = .{ .type = 0, .dev = 0, .qid = .{ .type = qtfile, .version = 0, .path = 0 }, .mode = 0, .atime = 0, .mtime = 0, .length = 0, .name = "", .uid = "", .gid = "", .muid = "", }; try testing.expectEqual(@as(u16, 47), try bare.size()); try testing.expectEqual(@as(usize, 49), (try bare.encode(&buf)).len); } test "9p: every message round-trips" { var buf: [512]u8 = undefined; _ = try roundTrip(&buf, notag, .{ .tversion = .{ .msize = 8192, .version = "9P2000" } }); _ = try roundTrip(&buf, notag, .{ .rversion = .{ .msize = 8192, .version = "9P2000" } }); // "unknown" is a SUCCESSFUL Rversion meaning no dialect in common, and the // codec must carry it like any other string rather than treat it as an // error path. _ = try roundTrip(&buf, notag, .{ .rversion = .{ .msize = min_msize, .version = "unknown" } }); _ = try roundTrip(&buf, 1, .{ .tauth = .{ .afid = 1, .uname = "goblin", .aname = "" } }); _ = try roundTrip(&buf, 1, .{ .rauth = .{ .aqid = .{ .type = qtauth, .version = 0, .path = 9 } } }); _ = try roundTrip(&buf, 2, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "goblin", .aname = "" } }); _ = try roundTrip(&buf, 2, .{ .rattach = .{ .qid = sample_qid } }); _ = try roundTrip(&buf, 3, .{ .rerror = .{ .ename = "no such file" } }); _ = try roundTrip(&buf, 4, .{ .tflush = .{ .oldtag = 3 } }); _ = try roundTrip(&buf, 4, .rflush); _ = try roundTrip(&buf, 5, .{ .twalk = .{ .fid = 0, .newfid = 1, .nwname = 2, .wname = .{ "7", "body" } ++ @as([max_welem - 2][]const u8, @splat("")) } }); _ = try roundTrip(&buf, 5, .{ .rwalk = .{ .nwqid = 2, .wqid = .{ sample_qid, sample_qid } ++ @as([max_welem - 2]Qid, @splat(sample_qid)) } }); _ = try roundTrip(&buf, 6, .{ .topen = .{ .fid = 1, .mode = 0 } }); _ = try roundTrip(&buf, 6, .{ .ropen = .{ .qid = sample_qid, .iounit = 8192 - iohdrsz } }); _ = try roundTrip(&buf, 7, .{ .tcreate = .{ .fid = 1, .name = "new", .perm = dmdir | 0o777, .mode = 2 } }); _ = try roundTrip(&buf, 7, .{ .rcreate = .{ .qid = sample_qid, .iounit = 0 } }); _ = try roundTrip(&buf, 8, .{ .tread = .{ .fid = 1, .offset = 0xdead_beef_cafe, .count = 4096 } }); _ = try roundTrip(&buf, 8, .{ .rread = .{ .data = "hello" } }); // A zero-byte Rread is end of file and not an error, which is exactly what // docs/9p.typ promises a pty reader on exit. _ = try roundTrip(&buf, 8, .{ .rread = .{ .data = "" } }); _ = try roundTrip(&buf, 9, .{ .twrite = .{ .fid = 1, .offset = 0, .data = "Edit ,d" } }); _ = try roundTrip(&buf, 9, .{ .twrite = .{ .fid = 1, .offset = 0, .data = "" } }); _ = try roundTrip(&buf, 9, .{ .rwrite = .{ .count = 7 } }); _ = try roundTrip(&buf, 10, .{ .tclunk = .{ .fid = 1 } }); _ = try roundTrip(&buf, 10, .rclunk); _ = try roundTrip(&buf, 11, .{ .tremove = .{ .fid = 1 } }); _ = try roundTrip(&buf, 11, .rremove); _ = try roundTrip(&buf, 12, .{ .tstat = .{ .fid = 1 } }); _ = try roundTrip(&buf, 12, .{ .rstat = .{ .stat = sample_stat } }); _ = try roundTrip(&buf, 13, .{ .twstat = .{ .fid = 1, .stat = sample_stat } }); _ = try roundTrip(&buf, 13, .rwstat); // Every type that has a message got one. The count is the thirteen pairs // plus Rerror; `Terror` is illegal and has no variant, which is what the // arithmetic below is really asserting. try testing.expectEqual(@as(usize, 27), @typeInfo(Msg).@"union".fields.len); try testing.expectEqual(std.enums.values(Type).len - 1, @typeInfo(Msg).@"union".fields.len); } test "9p: empty and maximum-length strings survive the trip" { var buf: [70_000]u8 = undefined; // Empty is not absent: the count is still two bytes. const empty = try roundTrip(&buf, 1, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "", .aname = "" } }); try testing.expectEqual(@as(usize, 0), empty.tattach.uname.len); try testing.expectEqual(@as(usize, header_len + 4 + 4 + 2 + 2), (try encode(empty, 1, &buf)).len); // The largest string a `n[2]` count can describe, and the one past it. var big: [65_536]u8 = undefined; @memset(&big, 'x'); const max = big[0..std.math.maxInt(u16)]; const got = try roundTrip(&buf, 1, .{ .rerror = .{ .ename = max } }); try testing.expectEqual(@as(usize, 65_535), got.rerror.ename.len); try testing.expectError(error.Overlong, encode(.{ .rerror = .{ .ename = &big } }, 1, &buf)); // ...and a stat whose strings overflow its own two-byte prefix. Refused by // `size()`, which is the only place that arithmetic happens. var wide = sample_stat; wide.name = max; try testing.expectError(error.Overlong, wide.size()); try testing.expectError(error.Overlong, encode(.{ .rstat = .{ .stat = wide } }, 1, &buf)); } test "9p: Twalk carries 0, 1 and 16 elements and refuses 17" { var buf: [512]u8 = undefined; // Zero elements is a legal walk and means "clone the fid". const zero = try roundTrip(&buf, 1, .{ .twalk = .{ .fid = 0, .newfid = 1, .nwname = 0 } }); try testing.expectEqual(@as(u16, 0), zero.twalk.nwname); try testing.expectEqual(@as(usize, header_len + 4 + 4 + 2), (try encode(zero, 1, &buf)).len); _ = try roundTrip(&buf, 1, .{ .twalk = .{ .fid = 0, .newfid = 1, .nwname = 1, .wname = .{"body"} ++ @as([max_welem - 1][]const u8, @splat("")), } }); // MAXWELEM exactly, all distinct so a swapped index cannot pass. const names: [max_welem][]const u8 = .{ "a", "b", "c", "d", "e", "f", "g", "h", "i", "j", "k", "l", "m", "n", "o", "p" }; const full = try roundTrip(&buf, 1, .{ .twalk = .{ .fid = 0, .newfid = 1, .nwname = max_welem, .wname = names } }); try testing.expectEqual(@as(u16, 16), full.twalk.nwname); for (names, full.twalk.wname[0..max_welem]) |a, b| try testing.expectEqualStrings(a, b); // Rwalk's bound is the same and its own. _ = try roundTrip(&buf, 1, .{ .rwalk = .{ .nwqid = max_welem, .wqid = @splat(sample_qid) } }); // Seventeen. Hand-built, because the encoder's array cannot hold one — the // point is that a PEER can send it and must be refused before the count // reaches an array of sixteen. var raw: [256]u8 = undefined; const bad = blk: { var w: Writer = .init(&raw); try w.putU32(0); // patched below try w.putByte(@intFromEnum(Type.twalk)); try w.putU16(1); try w.putU32(0); try w.putU32(1); try w.putU16(17); for (0..17) |i| try w.putString(&[_]u8{@intCast('a' + i)}); std.mem.writeInt(u32, raw[0..4], @intCast(w.n), .little); break :blk raw[0..w.n]; }; try testing.expectEqual(@as(usize, header_len + 4 + 4 + 2 + 17 * 3), bad.len); try testing.expectError(error.Overlong, decode(bad)); // Same for Rwalk: seventeen qids is 221 bytes of legal-looking message. const bad_r = blk: { var w: Writer = .init(&raw); try w.putU32(0); try w.putByte(@intFromEnum(Type.rwalk)); try w.putU16(1); try w.putU16(17); for (0..17) |_| try w.putQid(sample_qid); std.mem.writeInt(u32, raw[0..4], @intCast(w.n), .little); break :blk raw[0..w.n]; }; try testing.expectError(error.Overlong, decode(bad_r)); } test "9p: the stat double length" { var buf: [512]u8 = undefined; // THE fact. Rstat is `count[2]` then a record that begins with its own // `size[2]`, and the outer number is the inner one plus two — // `linux/net/9p/client.c:1633` reads it as "wS" and drops the first w. var good: [512]u8 = undefined; const n = blk: { const bytes = try encode(.{ .rstat = .{ .stat = sample_stat } }, 1, &buf); @memcpy(good[0..bytes.len], bytes); break :blk bytes.len; }; const inner = try sample_stat.size(); try testing.expectEqual(inner + 2, std.mem.readInt(u16, good[header_len..][0..2], .little)); try testing.expectEqual(inner, std.mem.readInt(u16, good[header_len + 2 ..][0..2], .little)); try testing.expectEqual(header_len + 2 + @as(usize, inner) + 2, n); // Twstat wraps the same pair behind a fid — `client.c:1776`, "dwS". const w_bytes = try encode(.{ .twstat = .{ .fid = 7, .stat = sample_stat } }, 1, &buf); try testing.expectEqual(inner + 2, std.mem.readInt(u16, w_bytes[header_len + 4 ..][0..2], .little)); try testing.expectEqual(inner, std.mem.readInt(u16, w_bytes[header_len + 6 ..][0..2], .little)); // Now three ways to get it wrong, which is the whole reason `Stat.decode` // is handed an exact slice instead of a cursor. Each is two bytes of edit // on a message that is otherwise perfect, and each is refused. var off: [512]u8 = undefined; // THE CLASSIC: the outer count written without the +2, so the record's own // prefix then claims two bytes more than the outer count allowed. @memcpy(off[0..n], good[0..n]); std.mem.writeInt(u16, off[header_len..][0..2], inner, .little); try testing.expectError(error.Truncated, decode(off[0..n])); // The outer count too large, which is the same mistake made twice. @memcpy(off[0..n], good[0..n]); std.mem.writeInt(u16, off[header_len..][0..2], inner + 4, .little); try testing.expectError(error.Truncated, decode(off[0..n])); // The INNER count wrong instead, in both directions: a record that claims // more than the outer count fits, and one that leaves bytes over inside it. @memcpy(off[0..n], good[0..n]); std.mem.writeInt(u16, off[header_len + 2 ..][0..2], inner + 2, .little); try testing.expectError(error.Truncated, decode(off[0..n])); @memcpy(off[0..n], good[0..n]); std.mem.writeInt(u16, off[header_len + 2 ..][0..2], inner - 1, .little); try testing.expectError(error.Trailing, decode(off[0..n])); } /// Every prefix of a complete message must be refused, in both of the two /// shapes a short message arrives in: /// * off a socket, where `size` still claims the whole thing and the header /// check catches it; /// * as a message that LIES about being complete, where `size` agrees with /// the buffer and only the per-field walk can catch it. This is the one /// that exercises every field boundary, and the one an attacker sends. /// Neither may panic and neither may parse. fn expectTruncatedAtEveryBoundary(full: []const u8) !void { var scratch: [1024]u8 = undefined; var n: usize = 0; while (n < full.len) : (n += 1) { try testing.expectError(error.Truncated, decode(full[0..n])); if (n < header_len) continue; @memcpy(scratch[0..n], full[0..n]); std.mem.writeInt(u32, scratch[0..4], @intCast(n), .little); try testing.expectError(error.Truncated, decode(scratch[0..n])); } // The complete message, by contrast, is fine — otherwise the loop above // would pass for a message that never decodes at all. _ = try decode(full); } test "9p: truncation at every field boundary is refused" { var buf: [512]u8 = undefined; // Tversion: size, type, tag, msize, a count, a string. try expectTruncatedAtEveryBoundary(try encode( .{ .tversion = .{ .msize = 8192, .version = "9P2000" } }, notag, &buf, )); // Twalk: two fids, a count, and then a loop of counted strings, which is // the only variable-arity field in the protocol. try expectTruncatedAtEveryBoundary(try encode(.{ .twalk = .{ .fid = 1, .newfid = 2, .nwname = 3, .wname = .{ "usr", "", "bin" } ++ @as([max_welem - 3][]const u8, @splat("")), } }, 1, &buf)); // Tread: the widest fixed body, and the one whose 8-byte offset a // native-struct blit would misalign. try expectTruncatedAtEveryBoundary(try encode( .{ .tread = .{ .fid = 1, .offset = 0x0102_0304_0506_0708, .count = 8168 } }, 1, &buf, )); // Rstat: both lengths, and every field of the record behind them. try expectTruncatedAtEveryBoundary(try encode(.{ .rstat = .{ .stat = sample_stat } }, 1, &buf)); // Rread, whose count is a u32 and whose payload is the message's tail. try expectTruncatedAtEveryBoundary(try encode(.{ .rread = .{ .data = "12345678" } }, 1, &buf)); // Rwalk, the other variable-arity body. try expectTruncatedAtEveryBoundary(try encode( .{ .rwalk = .{ .nwqid = 3, .wqid = @splat(sample_qid) } }, 1, &buf, )); // Twstat: a fid in front of the double length. try expectTruncatedAtEveryBoundary(try encode(.{ .twstat = .{ .fid = 1, .stat = sample_stat } }, 1, &buf)); } test "9p: a size field that disagrees with the buffer is refused" { var buf: [512]u8 = undefined; const bytes = try encode(.{ .tclunk = .{ .fid = 1 } }, 1, &buf); try testing.expectEqual(@as(usize, 11), bytes.len); var raw: [64]u8 = undefined; @memcpy(raw[0..bytes.len], bytes); // Larger than the buffer: not all of it has arrived. Every value up to a // hostile 4 GiB claim, which must not be believed for one instruction. for ([_]u32{ 12, 13, 64, 1 << 20, std.math.maxInt(u32) }) |claim| { std.mem.writeInt(u32, raw[0..4], claim, .little); try testing.expectError(error.Truncated, decode(raw[0..bytes.len])); } // Smaller than the buffer: two messages handed over as one. The caller's // framing is wrong, and silently decoding the first would hide it. std.mem.writeInt(u32, raw[0..4], 10, .little); try testing.expectError(error.Trailing, decode(raw[0..bytes.len])); // Smaller than a header at all: a number no encoder produced. Refused // before it is compared against the buffer, or a size of 3 on a 3-byte // buffer would slice a header out of nothing. for ([_]u32{ 0, 1, 6 }) |claim| { std.mem.writeInt(u32, raw[0..4], claim, .little); try testing.expectError(error.BadValue, decode(raw[0..bytes.len])); try testing.expectError(error.BadValue, decode(raw[0..header_len])); } } test "9p: an unknown or illegal type byte is refused" { var buf: [512]u8 = undefined; const bytes = try encode(.{ .tclunk = .{ .fid = 1 } }, 1, &buf); var raw: [64]u8 = undefined; @memcpy(raw[0..bytes.len], bytes); // 106 is `Terror`, defined and illegal. 8 is 9P2000.L's `Tstatfs`, 12 is // its `Tlopen`: dialects we do not serve, arriving as bytes we must refuse // rather than `@enumFromInt` into an exhaustive enum. for ([_]u8{ 0, 1, 8, 12, 99, 106, 128, 255 }) |t| { raw[4] = t; try testing.expectError(error.BadTag, decode(raw[0..bytes.len])); } // ...and the whole byte space, because the guarantee is total: a byte is // either a type we decode into a message of exactly that type, or an // error. Never a panic, and never a message of some OTHER type. var t: u16 = 0; while (t <= 255) : (t += 1) { raw[4] = @intCast(t); const defined = t >= 100 and t <= 127 and t != @intFromEnum(Type.terror); if (decode(raw[0..bytes.len])) |got| { try testing.expectEqual(@as(u8, @intCast(t)), @intFromEnum(got.msg.msgType())); // Exactly four types have a four-byte body: `fid[4]` for the three // T-messages and `count[4]` for Rwrite. Nothing else may decode // out of these bytes, and a fifth name here would mean a layout // above is wrong. try testing.expect(t == @intFromEnum(Type.tclunk) or t == @intFromEnum(Type.tremove) or t == @intFromEnum(Type.tstat) or t == @intFromEnum(Type.rwrite)); } else |err| { // An undefined byte, or the illegal 106, is ALWAYS BadTag: it must // never be diagnosed as a short body, because "read more" is the // wrong advice for a peer speaking another dialect. if (!defined) try testing.expectEqual(Error.BadTag, err); } } } test "9p: trailing bytes inside the size are refused" { var raw: [64]u8 = undefined; // A Tclunk whose `size` says twelve and whose body is five bytes: the fid // decodes, and one byte is left over. `convM2S.c:377-381` refuses the same // shape with `if(ap+size == p) return size; return 0;`. var w: Writer = .init(&raw); try w.putU32(12); try w.putByte(@intFromEnum(Type.tclunk)); try w.putU16(1); try w.putU32(7); try w.putByte(0xAA); try testing.expectEqual(@as(usize, 12), w.n); try testing.expectError(error.Trailing, decode(raw[0..12])); // Same for a body with room for a second copy of itself, which is how a // 9P2000.u message with an extra field would arrive. w = .init(&raw); try w.putU32(header_len + 2 + 2); try w.putByte(@intFromEnum(Type.tflush)); try w.putU16(1); try w.putU16(3); try w.putU16(3); try testing.expectError(error.Trailing, decode(raw[0..w.n])); } test "9p: frameLen needs four bytes" { var buf: [512]u8 = undefined; const bytes = try encode(.{ .tread = .{ .fid = 1, .offset = 0, .count = 8168 } }, 1, &buf); try testing.expectEqual(@as(usize, 23), bytes.len); // Zero through three: the reader has nothing to act on but "read more". for (0..4) |n| try testing.expectEqual(@as(?u32, null), frameLen(bytes[0..n])); // Four is enough, and the answer is the whole message including the four. try testing.expectEqual(@as(?u32, 23), frameLen(bytes[0..4])); try testing.expectEqual(@as(?u32, 23), frameLen(bytes)); // Unvalidated on purpose: the type byte may not have arrived yet, so there // is nothing to check the claim against. A caller compares it to its msize. var raw: [4]u8 = .{ 0xFF, 0xFF, 0xFF, 0xFF }; try testing.expectEqual(@as(?u32, std.math.maxInt(u32)), frameLen(&raw)); raw = .{ 0, 0, 0, 0 }; try testing.expectEqual(@as(?u32, 0), frameLen(&raw)); } test "9p: encode refuses a short buffer and writes nothing" { var buf: [512]u8 = undefined; const want = (try encode(.{ .rstat = .{ .stat = sample_stat } }, 1, &buf)).len; // Every buffer from empty to one byte short, because the interesting one is // not always the last: the message is sized before a byte is written, so // all of them must leave the buffer untouched. var n: usize = 0; while (n < want) : (n += 1) { var scratch: [512]u8 = @splat(0xAA); try testing.expectError(error.NoSpace, encode(.{ .rstat = .{ .stat = sample_stat } }, 1, scratch[0..n])); // NOTHING written, not even the size prefix — including past the end of // the slice it was given, which is the byte a length bug would reach. for (scratch) |b| try testing.expectEqual(@as(u8, 0xAA), b); } var exact: [512]u8 = @splat(0xAA); try testing.expectEqual(want, (try encode(.{ .rstat = .{ .stat = sample_stat } }, 1, exact[0..want])).len); try testing.expectEqual(@as(u8, 0xAA), exact[want]); } test "9p: byte for byte against u9fs convS2M" { var buf: [512]u8 = undefined; // A round-trip test proves the two halves of THIS file agree. These four // prove they agree with the reference implementation, which is what // plan9port's `9p`, Plan 9's mount driver and Linux's v9fs are all // compatible with. Each array was written out by hand from `convS2M.c` and // the line is named. // Tversion, `convS2M.c:236-240` with the header at :224-229. // size[4]=19 type[1]=100 tag[2]=NOTAG msize[4]=8192 version[2+6] // 19, not 12: `size` counts itself and the type and the tag. try testing.expectEqualSlices(u8, &.{ 0x13, 0x00, 0x00, 0x00, // size = 19 0x64, // Tversion = 100 0xff, 0xff, // NOTAG 0x00, 0x20, 0x00, 0x00, // msize = 8192 0x06, 0x00, // n = 6 '9', 'P', '2', '0', '0', '0', }, try encode(.{ .tversion = .{ .msize = 8192, .version = "9P2000" } }, notag, &buf)); // Twalk, `convS2M.c:272-283`: fid, newfid, nwname, then `pstring` each, // and `pstring` (:4-16) writes `n[2]` with NO terminator. try testing.expectEqualSlices(u8, &.{ 0x1b, 0x00, 0x00, 0x00, // size = 27 0x6e, // Twalk = 110 0x01, 0x00, // tag = 1 0x01, 0x00, 0x00, 0x00, // fid = 1 0x02, 0x00, 0x00, 0x00, // newfid = 2 0x02, 0x00, // nwname = 2 0x03, 0x00, 'u', 's', 'r', 0x03, 0x00, 'b', 'i', 'n', }, try encode(.{ .twalk = .{ .fid = 1, .newfid = 2, .nwname = 2, .wname = .{ "usr", "bin" } ++ @as([max_welem - 2][]const u8, @splat("")), } }, 1, &buf)); // Rread, `convS2M.c:392-397`: count[4] then the bytes. The 11-byte header // this implies is where `msize - 11` comes from (docs/registry.typ). try testing.expectEqualSlices(u8, &.{ 0x0e, 0x00, 0x00, 0x00, // size = 14 0x75, // Rread = 117 0x09, 0x00, // tag = 9 0x03, 0x00, 0x00, 0x00, // count = 3 'a', 'b', 'c', }, try encode(.{ .rread = .{ .data = "abc" } }, 9, &buf)); // Rstat, `convS2M.c:410-415` (`PBIT16(p, f->nstat)` then the blob) around // `convD2M.c:50-83` (the record, whose own prefix is `ss - BIT16SZ`). THE // double length, in bytes: 53 outside, 51 inside, 55 of body, 62 total. const one: Stat = .{ .type = 0, .dev = 0, .qid = .{ .type = qtdir, .version = 1, .path = 2 }, .mode = dmdir | 0o755, .atime = 3, .mtime = 4, .length = 0, .name = "a", .uid = "u", .gid = "g", .muid = "m", }; try testing.expectEqual(@as(u16, 51), try one.size()); try testing.expectEqualSlices(u8, &.{ 0x3e, 0x00, 0x00, 0x00, // size = 62 0x7d, // Rstat = 125 0x07, 0x00, // tag = 7 0x35, 0x00, // OUTER count = 53 = 51 + 2 0x33, 0x00, // stat size = 51, excluding these two 0x00, 0x00, // type 0x00, 0x00, 0x00, 0x00, // dev 0x80, // qid.type = QTDIR 0x01, 0x00, 0x00, 0x00, // qid.version = 1 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // qid.path = 2 0xed, 0x01, 0x00, 0x80, // mode = DMDIR | 0755 0x03, 0x00, 0x00, 0x00, // atime 0x04, 0x00, 0x00, 0x00, // mtime 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // length 0x01, 0x00, 'a', // name 0x01, 0x00, 'u', // uid 0x01, 0x00, 'g', // gid 0x01, 0x00, 'm', // muid }, try encode(.{ .rstat = .{ .stat = one } }, 7, &buf)); // The high five mode bits ARE the qid type bits, shifted down 24 // (`protocol.rs:486-495`). Asserted here rather than implemented, because // this codec carries both fields and the server half sets them. try testing.expectEqual(qtdir, @as(u8, @intCast(dmdir >> 24))); try testing.expectEqual(qtappend, @as(u8, @intCast(dmappend >> 24))); try testing.expectEqual(qtexcl, @as(u8, @intCast(dmexcl >> 24))); try testing.expectEqual(qtauth, @as(u8, @intCast(dmauth >> 24))); try testing.expectEqual(qttmp, @as(u8, @intCast(dmtmp >> 24))); try testing.expectEqual(@as(u32, 0o777), dmperm); } // =========================================================================== // THE SERVER HALF // =========================================================================== // // Everything above is the wire. Everything below turns a stream of those // messages into `acmefs.Req` and back. // // IT IS A SANS-IO STATE MACHINE and it never touches a descriptor: the caller // pushes bytes in with `push`, pumps requests through the core with // `retry`/`next`/`reply`, and takes bytes out with `output`/`wrote`. There is // no socket here, no poll, no thread and no allocator, which is the whole // point — the same code serves a unix socket on Linux, a TCP connection from // another machine, and the board's UART, and the transport-specific part is // two syscalls in the caller. // // WHAT IT IS NOT. It is not a filesystem: every question about what a file // MEANS belongs to `acmefs.zig`, and this half knows only that a node id is a // u64, that some nodes are directories, and that a reply may say "ask me // later". There are exactly two exceptions, both named and both forced by the // absence of a kernel: the root's node id, which arrives as a parameter to // `init`, and `parentOf`, which is where `..` goes. // // Verified against `u9fs/u9fs.c` (the tree and the offset rules), // `linux/net/9p/{client,error}.c` (what a real client does with our answers), // `principia-softwarica/lib_networking/lib9p/srv.c` (flush ordering) and // `ad/crates/ninep/src/sansio/server.rs` (the scars in its git history). // --------------------------------------------------------------------------- // the error ABI // --------------------------------------------------------------------------- // // `Rerror` carries a STRING and base 9P2000 has no number beside it, so the // WORDING IS THIS SERVER'S ERROR ABI. Linux recovers an errno by exact match // against a fixed table and a miss is not `EIO` but `ESERVERFAULT`, which // userspace prints as "Unknown error 526" — so every string below is copied // character for character out of `linux/net/9p/error.c:41-171`, with the errno // it maps to named beside it. // // THIS WAS A CHOICE and `docs/registry.typ` `9P-4` left it open with three // candidates. This is OPTION A. Option B was to serve 9P2000.u and send the // number, which also restores `Tstatfs` — the one `acmefs.Op` with no // base-9P2000 message — and keeps a human-readable string for Plan 9 clients; // it costs a second dialect inside every parser above, because `.u` changes // the LAYOUT of `Rerror` and `stat` rather than adding messages. Option C was // acme's own wording (`Ebadctl`, `Ebadaddr`, `Ebadevent`), which is a table // miss for every one of them; that is what `ad` shipped, so under // `mount -t 9p` every error it can produce arrives as 526. // // The cost of option A is stated plainly: English kernel strings become this // project's error ABI, and a script reading `event` sees "Invalid argument" // where acme would have said something about a read too small. If `.u` is ever // served this table stays as it is — `.u`'s `Rerror` carries the string too. /// EBADF. The fid a message names was never walked to, or has been clunked. /// u9fs spells it `Ebadfid` (`u9fs.c:119`). pub const e_unknown_fid = "fid unknown or out of range"; /// EBADF. `Tattach` or `Twalk` named a `newfid` that is already bound. u9fs /// `Efidactive` (`u9fs.c:124`). pub const e_fid_in_use = "fid already in use"; /// EBADF. A fid used for something its state does not allow: read on a fid /// that was never opened, write on one opened `OREAD`, walk from an open one. /// u9fs `Ebadusefid` (`u9fs.c:121`), whose five uses are ours. pub const e_bad_use = "bad use of fid"; /// ESPIPE. A directory read at an offset that is neither zero nor exactly /// where the last one ended. u9fs `Ebadoffset` (`u9fs.c:763`). pub const e_bad_offset = "bad offset in directory read"; /// EACCES. The permission bits the core reported do not admit this open, or /// the message asked for something a generated tree cannot do: create, remove, /// remove-on-close, execute. pub const e_perm = "permission denied"; /// ENOTDIR. A walk with names from a fid that is not a directory. pub const e_not_dir = "not a directory"; /// ETXTBSY. A second `Topen` on one fid. The fid IS the open, so there is /// nothing for the second one to mean. pub const e_already_open = "file already open for I/O"; /// ENAMETOOLONG. A walk element longer than `name_max`. No name in this tree /// is, so this is a client asking for something that cannot exist — refused on /// its length rather than looked up, because the fid has to be able to hold /// the name it lands on (`Rstat` carries it). pub const e_illegal_name = "illegal name"; /// ENFILE. The fid table is full. Thirty-two is a lot of scripts. pub const e_too_many_fids = "Too many open files in system"; /// EPROTO. Not 9P2000 on this connection: an R-message arriving at a server, a /// type byte no dialect we serve defines, a body that does not parse, a /// message larger than the negotiated msize, or anything at all before /// `Tversion`. pub const e_botch = "protocol botch"; /// EINTR. What a flushed request is answered with, immediately before its /// `Rflush`. The same answer `fuse.zig:1338` gives a `FUSE_INTERRUPT`. pub const e_interrupted = "Interrupted system call"; /// EPERM. A `Twstat` carrying a non-zero length. The core honours exactly one /// field and only the value zero (`acmefs.zig:1096-1104`), and this string /// says so in a wording Linux already knows. pub const e_trunc_only = "only support truncation to zero length"; /// EPERM. A `Twstat` that would rename, or otherwise change the shape of a /// tree that follows the pane list. pub const e_wstat = "wstat prohibited"; /// EAGAIN. The park table is full, or the core parked a request whose payload /// is too large to copy into a slot. Both are honest to a client: retry. pub const e_again = "Resource temporarily unavailable"; /// EINVAL. A read whose count cannot hold the first thing the answer consists /// of — one directory entry. Refused rather than answered short, because a /// `Tread` returning zero bytes is END OF DIRECTORY and a client that believes /// it stops asking. The same rule `acmefs` already applies to an `event` read /// too small for one record, and `9P-17` records that this is what makes the /// clamp to `count` safe. pub const e_count_small = "Invalid argument"; /// ENOENT. `Tattach` named an `aname`. There is one tree here and it has no /// name; a client that asked for a different one should learn that now rather /// than be handed this one (docs/9p.typ §12.4: "no `aname`"). pub const e_no_tree = "No such file or directory"; /// Not in Linux's table, and deliberately: Linux's client never sends `Tauth` /// at all, and Plan 9's `mount` treats an error here as "no authentication /// needed" and carries on. So this string is chosen for the human reading a /// Plan 9 error message rather than for `p9_errstr2errno`. u9fs says the same /// thing in a string that maps to zero — "not an error" — which is a subtlety /// we do not need. Real authentication is `Tauth` or a tunnel, and neither is /// ours (docs/9p.typ §10). pub const e_no_auth = "authentication not required"; /// EINVAL. `Tversion` offered an msize too small to serve (see `msize_min`). /// `Rversion` has no way to say this — its `version` field means "no dialect /// in common", which is a different fact — and `Rerror` is a legal reply to /// any T-message, so this is the honest channel. pub const e_small_msize = "Invalid argument"; /// The core's numeric errno as the string Linux turns back into that same /// number. Every value `acmefs.E` defines is here by name; anything else /// becomes EIO, because a number we did not choose to emit is a bug in this /// file and "Input/output error" is the one answer that is never misleading. pub fn errString(errno: u16) []const u8 { return switch (errno) { 1 => "Operation not permitted", // E.PERM, EPERM 2 => "No such file or directory", // E.NOENT, ENOENT 5 => "Input/output error", // E.IO, EIO 12 => "Cannot allocate memory", // E.NOMEM, ENOMEM 20 => "Not a directory", // E.NOTDIR, ENOTDIR 22 => "Invalid argument", // E.INVAL, EINVAL 23 => "Too many open files in system", // E.NFILE, ENFILE 28 => "No space left on device", // E.NOSPC, ENOSPC 38 => "Function not implemented", // E.NOSYS, ENOSYS else => "Input/output error", }; } // --------------------------------------------------------------------------- // open modes // --------------------------------------------------------------------------- // // `Topen.mode`, from `u9fs/plan9.h:146-153`. The codec above carries the byte // and has no opinion about it; these are what the byte MEANS, which is the // server's business. /// The low two bits, which are a VALUE and not a mask: 0, 1, 2, 3. pub const oread: u8 = 0; pub const owrite: u8 = 1; pub const ordwr: u8 = 2; /// «execute, == read but check execute permission». Nothing in this tree is a /// program, so it is refused rather than treated as a read. pub const oexec: u8 = 3; /// Or'ed in. Truncate first — this is how a shell's `>` reaches a 9P server, /// and it maps onto `acmefs.Req.truncate` exactly as a `Twstat` with a zero /// length does (docs/registry.typ `FIX-1`). pub const otrunc: u8 = 16; /// Or'ed in, close on exec. A CLIENT-SIDE flag: Plan 9's kernel consumes it /// and never sends it, so a server that sees it may ignore it, and we do. pub const ocexec: u8 = 32; /// Or'ed in, remove on close. Refused: this tree's shape follows the pane list /// and there is nothing in it a client may remove. pub const orclose: u8 = 64; // --------------------------------------------------------------------------- // sizes // --------------------------------------------------------------------------- /// Fids one connection may hold at once. A FIXED ARRAY and not a map, costed /// in `docs/registry.typ` `9P-11`: a linear scan is far cheaper than the wire, /// so the map would buy nothing and cost an allocator this file does not have. /// /// 256 AND NOT 32, which is what it was, and the difference is a MOUNT. A /// script that opens one file at a time never needs more than a handful; a /// mounting client keeps one fid per cached inode, and this tree is three /// top-level entries plus fourteen files per pane, so seven panes already pass /// thirty-two and a full sixteen-pane session wants over two hundred. At the /// old number `find` over a `9pfuse` mount failed with fifty-seven consecutive /// `Rerror`s once the table filled — and `9pfuse` is the proof clause /// `docs/9p.typ` §12.4 sets for this step, so the number was refuting its own /// acceptance test. /// /// A `Fid` is about 64 bytes, so this is ≈16 KiB per connection against the /// ≈34 KiB `fs9_service.zig` already budgets for one. The BOARD keeps thirty-two /// by passing its own value: see `board_fids`, and `9P-11`'s RAM line, which is /// costed for a microcontroller serving its own small tree and nothing else. /// /// Overflow is a refusal (`e_too_many_fids`), not a queue. pub const max_fids: usize = 256; /// What a microcontroller uses instead. Named here rather than spelled at the /// call site so that the two numbers, and the reason they differ, stay next to /// each other. pub const board_fids: usize = 32; /// Requests that may be outstanding at once — in flight, or parked because the /// core answered `.again`. In practice this counts BLOCKED READERS: one slot /// per process sitting on `event`. The number is `src/fuse.zig:793`'s, /// unchanged, because `Status.again` means the same thing to both transports. pub const max_slots: usize = 32; /// Bytes of request payload a park slot owns. A parked request's `data` cannot /// go on borrowing the input buffer — the next message overwrites it — so it /// is copied in when it fits. /// /// Smaller than `fuse.zig`'s 512, for a reason specific to this file: under /// FUSE a LOOKUP name is a payload and may be 255 bytes, while a 9P walk /// element is consumed inside the walk and never parks. What is left is a /// write, and the only writes that could conceivably block are a `ctl` verb /// line and an event write-back, both a few dozen bytes. A larger write that /// the core tries to park is answered `e_again` — honest, and by construction /// unreachable, since the core answers writes as transactions. pub const park_data_max: usize = 128; /// The longest name a fid may land on, and the longest `uname` we keep. /// /// A BOUND rather than a buffer size: `Rstat` carries the file's name, so a /// fid has to hold the name it walked to, and this is the number that makes /// `msize_min` provable. Every name in the tree fits with room over — the /// longest are a pane's decimal serial and `errors` — so a walk element longer /// than this is refused as `e_illegal_name` rather than looked up and then /// truncated, which would make `Rstat` lie. pub const name_max: usize = 28; /// The smallest msize this server will agree to serve. DERIVED, not chosen: /// `Rwalk` with the protocol's sixteen qids is the largest reply whose size /// the client cannot influence after the handshake, so a connection that /// cannot hold one cannot be served at all. /// /// Deliberately NOT `min_msize` (4096), which is the LINUX KERNEL's floor and /// nobody else's: Plan 9's devmnt, plan9port's `9p` and pardes's own client all /// accept 512, and the board would rather have the kilobytes back. pub const msize_min: u32 = header_len + 2 + max_welem * qid_len; comptime { assert(msize_min == 217); // The other two replies whose size the client does not choose: `Rstat` // carries one record with four strings, and a directory read must fit at // least one such record or it can never make progress. Both must clear // `msize_min`, or the floor above is not a floor. assert(header_len + 2 + stat_fixed + 4 * name_max <= msize_min); assert(header_len + 4 + stat_fixed + 4 * name_max <= msize_min); // A pane serial is a u60, so its decimal name is at most twenty digits and // `parentOf` cannot overflow the buffer it formats into. assert(name_max >= 20); // Modes are a value in the low two bits with flags above them. assert(oread | owrite | ordwr | oexec == 3); assert(otrunc | ocexec | orclose == 112); } /// The server's name for a node. /// /// `qid.version` IS ALWAYS ZERO, and this is a policy rather than a /// translation. It is the 9P equivalent of the `FOPEN_DIRECT_IO` that /// `src/fuse.zig:172-176` relies on, and it is server-side rather than advice /// to whoever mounts: Linux's client sets `P9L_DIRECT` — «no read or write /// cache» — for any file whose qid version is zero, whatever the cache mode, /// unless `ignoreqv` is passed explicitly (`linux/fs/9p/fid.h:52-53`, /// `v9fs.c:93`). A synthetic tree of live editor state has no business being /// cached: `body` changes under the reader's feet, `event` is a queue, and a /// cached lookup under `new/` would create one pane and then serve the same /// answer forever. Plan 9 needs nothing said to it — its cache is opt-in via /// `mount -c` (docs/registry.typ `9P-3`). /// /// `qid.path` is the core's node id UNCHANGED, which is what makes the two /// transports agree: one integer is a FUSE nodeid, a `d_ino` and a qid path at /// once, and it never comes to mean a different file because pane serials are /// never reused (`acmefs.zig:231-237`). fn qidOf(node: u64, dir: bool) Qid { return .{ .type = if (dir) qtdir else qtfile, .version = 0, .path = node }; } /// Where `..` goes, and the ONE place in this file that decodes a node id. /// /// WHY THIS IS HERE AT ALL, because it is the fact that gets lost: under FUSE /// the kernel resolves `.` and `..` in the pathname before a request is ever /// sent, which is what lets `acmefs.zig:840` say they «are the kernel's /// business, never ours». Under 9P THERE IS NO KERNEL. `Twalk` carries `..` as /// an ordinary name element, and a client that normalises a path, or walks up /// before walking down, sends it. Forwarding it to the core as a lookup would /// answer `ENOENT` and break `cd ..`, so the server answers it. /// /// It can, without asking anything, because the tree has fixed depth and the /// node id says where you are. `acmefs.Node` is /// `packed struct(u64){ file: u4, serial: u60 }` (`acmefs.zig:238-240`), so /// `file` is the low four bits and `serial` is everything above them, and: /// /// * at the root, `..` is the root. POSIX's rule and `intro(5)`'s: the root /// is its own parent, and this is not an error. /// * `serial == 0` is a top-level file (`index`, `cons`, `new`), whose /// parent is the root. /// * `file == 0` is `PaneFile.dir`, a pane's own directory, whose parent is /// the root. /// * anything else is a file inside a pane's directory, and its parent is /// that directory: the same serial with `file` cleared. Its NAME is the /// serial in decimal, which is how `acmefs`'s root lists it /// (`acmefs.zig:992-994`). /// /// A well-behaved client only walks between directories, so the last case /// should never arrive; it is answered correctly rather than trusted away. /// /// THE DEPTH ASSUMPTION IS THE WHOLE OF WHAT COULD ROT, and it is checked by /// the shape of the node id rather than by hope: `Node` has ONE `file: u4`, so /// a level below a pane's directory — `docs/9p.typ`'s `pty/` — cannot be /// encoded in a node id at all today. If that changes, this function is the /// one place that has to learn about it. fn parentOf(node: u64, root: u64, buf: *[name_max]u8) struct { node: u64, name_len: u8 } { const serial = node >> 4; const file = node & 0xF; if (node == root or serial == 0 or file == 0) { buf[0] = '/'; return .{ .node = root, .name_len = 1 }; } // A u60 is twenty decimal digits at most and `name_max` is checked against // that above, so the format cannot fail. const name = std.fmt.bufPrint(buf, "{d}", .{serial}) catch unreachable; return .{ .node = serial << 4, .name_len = @intCast(name.len) }; } /// The permission bits a DIRECTORY ENTRY reports, and the only place in this /// file that reports a mode it was not told. /// /// `acmefs`'s staging format for a readdir is `node[8] dir[1] namelen[1] /// name[]` (`acmefs.zig:942-957`) — the same record the FUSE transport decodes /// — and it carries no mode and no length, because under FUSE the kernel asks /// for those separately, with a `getattr` per entry it decides it wants. 9P /// puts a whole `stat` in a directory read, so the choice is between a /// `getattr` per entry — an extra round trip each, and a third msize buffer to /// hold the entries across it — and reporting the tree's own defaults here. /// /// We report the defaults. `0o500` is what `TopFile.mode` and `PaneFile.mode` /// give every directory in the tree without exception; `0o600` is what /// `PaneFile.mode` gives every file but three; a length of zero is the true /// length of every file here but `body`, `tag` and `index`. /// /// WHO SEES THE DIFFERENCE: only a client that reads permissions and sizes out /// of a DIRECTORY READ, which is Plan 9's `ls -l` and nothing else. Linux's /// v9fs takes names and qids from the read and stats each file separately, /// 9pfuse does the same, and `Tstat` here answers out of the core's own /// `getattr` — so `ls -l` through either of those is exact. pub const dirent_dir_perm: u16 = 0o500; pub const dirent_file_perm: u16 = 0o600; /// A 9P2000 server for one connection, over the filesystem ABI `fs`. /// /// WHY THIS IS A GENERIC and not a plain struct that imports `acmefs.zig`: /// this file is freestanding-safe and must stay so — it compiles for /// `wasm32-freestanding` and the board's `riscv32-freestanding`, and /// `acmefs.zig` reaches `pardes.zig`, which reaches the build's generated /// modules. Importing it would also drag every test in that graph into /// `zig test src/9p.zig`. So the ABI arrives as a type parameter and the /// coupling is exactly three declarations: /// /// * `fs.Req` with `tag, op, node, handle, off, size, data, truncate` /// * `fs.Reply` with `tag, status, errno, attr, handle, written` /// * `fs.Reply.Attr` with `node, dir, size, mode` /// /// which is `acmefs`'s ABI verbatim, so the real instantiation is /// `Server(acmefs)` and it needs no translation layer at all. The `Op` and /// `Status` values are reached as enum literals (`.lookup`, `.again`), so they /// are checked against the real enums at that instantiation. The tests below /// instantiate it on a stub filesystem, which is how they run with no core. /// /// THE THREE METHODS `src/fs_service.zig`'s `Transport` wants — `retry`, /// `next` and `reply` — are here with those names and those shapes, and the /// order contract is that file's: `retry()` to null first, then `next()` to /// null. `fs_service` is deliberately NOT imported (it is `std.c` and /// `pardes.zig` deep); the adapter that fills in a vtable is three functions /// in whoever owns the socket. /// /// MEMORY, all of it caller-supplied or fixed: the two buffers, a fid table of /// `max_fids` and a park table of `max_slots`. No allocator, and nothing here /// grows. pub fn Server(comptime fs: type) type { return struct { const Self = @This(); /// Bytes the caller has pushed and we have not finished with. /// `in[0..frame]` is the message being served when `frame != 0`, and /// every slice a decoded `Msg` holds points into it — which is why /// nothing compacts this buffer until that message is done with. in: []u8, /// Encoded replies, oldest first, as a byte FIFO. Every 9P message /// carries its own length, so the queue needs no side table: the /// caller writes `output()` and tells us how much went. out: []u8, /// The node id of the tree's root — `@intFromEnum(acmefs.TopFile.root)` /// — and the one fact about the tree this file is told rather than /// deriving. `Tattach` needs somewhere to start and 9P has no way to /// ask for it. root: u64, in_len: usize = 0, frame: u32 = 0, out_len: usize = 0, out_off: usize = 0, /// Negotiated by `Tversion`; ZERO means not yet, and nothing but /// `Tversion` is served in that state. msize: u32 = 0, /// The stream is not 9P and there is no resynchronising from it: stop /// serving and let the caller close. Write-once, like `fuse.Fs.dead`. dead: bool = false, /// Whoever attached, for `Rstat`'s three name fields. The tree is /// synthetic and has one owner: the client that opened the connection. uname: [name_max]u8 = @splat(0), uname_len: u8 = 0, fids: [max_fids]Fid = @splat(.{}), slots: [max_slots]Slot = @splat(.{}), /// The message being served. At most one, which is what keeps the /// walk's accumulated qids and the borrowed names in one place instead /// of in thirty-two slots. job: Job = .{}, /// Hands out `fs.Req.tag`s, and orders the park table. Never zero, so /// that zero can mean "no request outstanding". seq: u64 = 0, /// What a 9P message is being turned into. The reply's SHAPE, which is /// what `reply` needs and what `Op` alone does not say: a `getattr` is /// a step of `Rattach`, of `Rwalk` and of `Rstat`. const Kind = enum { none, attach, walk, open, read, readdir, write, clunk, remove, stat, wstat }; /// One fid: a name the client gave a place in the tree. /// /// `perm` and `dir` are cached from the attributes the walk that landed /// here already answered, because `Topen` has to check permission /// itself — there is no kernel above us doing it, and `acmefs.open` /// deliberately does not (`acmefs.zig:1023-1031`). `name` is cached /// because `Rstat` carries it and a node id does not. const Fid = struct { used: bool = false, /// The client's number. `nofid` is never one. fid: u32 = 0, node: u64 = 0, dir: bool = false, /// Permission bits as the core last reported them, which is what /// `Topen` is checked against. perm: u16 = 0, open: bool = false, /// The `Topen` mode, valid when `open`. omode: u8 = 0, /// `acmefs`'s open handle, repeated on every read, write and /// release. handle: u32 = 0, /// THE DIRECTORY CURSOR, in the two coordinate systems it has to /// live in at once: `diroff` is the BYTE offset 9P requires the /// next read to carry, and `dirindex` is the ENTRY INDEX `acmefs` /// counts in (`acmefs.zig:972`, `var skip = req.off;`). diroff: u64 = 0, dirindex: u32 = 0, /// This fid has no client any more and still owes the core a /// `release`. See `orphan`. orphan: bool = false, name: [name_max]u8 = @splat(0), name_len: u8 = 0, }; /// A request the core would not answer yet. Lifted from /// `src/fuse.zig:806-822` with the FUSE opcode replaced by the 9P tag /// and the reply shape, because `Status.again` means the same thing to /// both transports and this is where `docs/registry.typ` `9P-16` says /// we beat the prior art. const Slot = struct { used: bool = false, /// The core answered `.again`; `retry()` will offer it back. parked: bool = false, /// Already offered in this retry round. Reset when a round finds /// nothing, which gives every parked request exactly one attempt /// per frame instead of letting the oldest starve the rest. retried: bool = false, /// `req.data` points into `data` below rather than into `in`. copied: bool = false, /// Arrival order, so retries are FIFO: the reader that blocked /// first is offered first. seq: u64 = 0, /// The client's tag, which is what `Tflush` names. tag: u16 = 0, kind: Kind = .none, /// The client's fid NUMBER and not an index: the fid may be /// clunked while this is parked, and a stale index would be a /// stale pointer. fid: u32 = 0, /// What the client asked for, which is what the answer is clamped /// to (`docs/registry.typ` `9P-17`). count: u32 = 0, req: fs.Req = undefined, data: [park_data_max]u8 = undefined, }; /// The message in flight, and the accumulated answer. const Job = struct { kind: Kind = .none, tag: u16 = 0, /// The `fs.Req.tag` of the step the core is holding, or zero. req_tag: u64 = 0, /// The step itself, kept so that a park has something to copy and /// a retry has something to re-offer. req: fs.Req = undefined, step: u8 = 0, fid: u32 = 0, newfid: u32 = 0, count: u32 = 0, offset: u64 = 0, omode: u8 = 0, /// Where the walk has got to: the node, its attributes and its /// name, all of which `Rwalk`'s last qid and the bound fid need. node: u64 = 0, dir: bool = false, perm: u16 = 0, name: [name_max]u8 = @splat(0), name_len: u8 = 0, nwname: u8 = 0, nwqid: u8 = 0, wqid: [max_welem]Qid = @splat(.{ .type = 0, .version = 0, .path = 0 }), /// The decoded T-message, BORROWING `in[0..frame]`: a walk's names /// and a write's bytes live here and nowhere else. msg: Msg = .rflush, }; pub const Options = struct { /// Room for one whole T-message. Caps the msize we will agree to, /// with `out`. in: []u8, /// Room for two: one being written out and one being built. That /// is what lets a reply be encoded the moment the core answers, /// with no "can I write yet" question anywhere in this file. out: []u8, /// `@intFromEnum(acmefs.TopFile.root)`. root: u64, }; /// The buffers are the caller's, which is what "no allocator" means /// here: the board hands over two static arrays, a desktop host hands /// over two heap slices sized for a 128 KiB msize, and this file cannot /// tell the difference. The msize follows from them and from the /// client's `Tversion`; see `version`. pub fn init(opts: Options) Self { assert(opts.in.len >= msize_min); assert(opts.out.len >= 2 * msize_min); // Node zero is `acmefs.Node{}` — no file, no pane — and cannot be // a root. A zero here would make every `..` land on nothing. assert(opts.root != 0); return .{ .in = opts.in, .out = opts.out, .root = opts.root }; } /// The connection went away. Every open fid still owes the core a /// `release`, and that debt outlives the connection: an `event` fid /// dropped without one leaves the pane's reader count high forever, /// which leaves the editor reporting button actions to a script that /// is no longer there (`acmefs.zig:1053-1061`). So the fids are /// ORPHANED rather than forgotten, and the caller keeps pumping /// `next()` until it answers null. pub fn hangup(s: *Self) void { s.reset(); s.dead = true; s.in_len = 0; s.frame = 0; s.out_len = 0; s.out_off = 0; } /// What `Tversion` does to the connection, and what `hangup` does /// first: «all fids are clunked and any outstanding I/O is abandoned» /// (`version(5)`). The parked requests go without an answer, which is /// exactly what abandoned means; the fids that are open become /// orphans, because the core's side of an open is not the client's to /// abandon. fn reset(s: *Self) void { for (&s.fids) |*f| { if (!f.used) continue; if (f.open) f.orphan = true else f.* = .{}; } for (&s.slots) |*sl| sl.* = .{}; s.job = .{}; } // -- bytes in, bytes out --------------------------------------------- /// Take as much of `bytes` as there is room for, and answer how much. /// A short answer is not an error and not a loss: it is the only /// back-pressure a sans-io server has, and the caller re-offers the /// tail after pumping. Bytes are APPENDED, so a message already being /// served does not move. pub fn push(s: *Self, bytes: []const u8) usize { if (s.dead) return 0; const n = @min(bytes.len, s.in.len - s.in_len); @memcpy(s.in[s.in_len..][0..n], bytes[0..n]); s.in_len += n; return n; } /// The replies waiting to go, oldest first, as one contiguous run of /// whole 9P messages. Valid until the next call to anything else here. pub fn output(s: *const Self) []const u8 { return s.out[s.out_off..s.out_len]; } /// How many of `output()`'s bytes actually left. A partial write is /// normal on a UART and on a full socket, and the remainder stays put. pub fn wrote(s: *Self, n: usize) void { assert(n <= s.out_len - s.out_off); s.out_off += n; if (s.out_off == s.out_len) { s.out_off = 0; s.out_len = 0; } } /// Slide the unwritten tail down. Called only when room is wanted, so /// the common case — a fully written queue, reset to empty by `wrote` — /// never moves a byte. fn compact(s: *Self) void { assert(s.out_off <= s.out_len); const n = s.out_len - s.out_off; std.mem.copyForwards(u8, s.out[0..n], s.out[s.out_off..s.out_len]); s.out_off = 0; s.out_len = n; } /// THE RESERVATION RULE, and the reason no reply in this file can ever /// fail to be written: a request is not handed to the core unless the /// out queue already has room for the largest answer it could produce, /// which is one msize. So `emit` cannot run out, a parked read that /// completes cannot be dropped, and back-pressure lands where it can /// be dealt with — `next()` and `retry()` answer null, the caller /// writes some bytes, and the pump continues. fn hasRoom(s: *Self) bool { if (s.out_off != 0) s.compact(); return s.out.len - s.out_len >= @max(s.msize, msize_min); } /// Queue one reply. Infallible by the reservation rule above; if it /// ever is not, the connection dies rather than the stream growing a /// half-written message — a dropped reply hangs a client forever, /// while a closed connection makes it fail and say so. fn emit(s: *Self, tag: u16, msg: Msg) void { const bytes = encode(msg, tag, s.out[s.out_len..]) catch { s.dead = true; return; }; s.out_len += bytes.len; } fn fail(s: *Self, tag: u16, ename: []const u8) void { assert(ename.len <= errmax); s.emit(tag, .{ .rerror = .{ .ename = ename } }); } /// The next `fs.Req.tag`. Unique for the life of the connection, which /// is what lets `reply` find its target with no cooperation from the /// core, and never zero. fn tick(s: *Self) u64 { s.seq += 1; return s.seq; } fn findFid(s: *Self, fid: u32) ?usize { for (&s.fids, 0..) |*f, i| if (f.used and !f.orphan and f.fid == fid) return i; return null; } fn freeFid(s: *Self) ?usize { for (&s.fids, 0..) |*f, i| if (!f.used) return i; return null; } fn dropFid(s: *Self, fid: u32) void { if (s.findFid(fid)) |i| s.fids[i] = .{}; } fn findSlot(s: *Self, req_tag: u64) ?usize { for (&s.slots, 0..) |*sl, i| if (sl.used and sl.req.tag == req_tag) return i; return null; } fn freeSlot(s: *Self) ?usize { for (&s.slots, 0..) |*sl, i| if (!sl.used) return i; return null; } /// A parked request by the tag the CLIENT gave it, which is what /// `Tflush` names. fn findTag(s: *Self, tag: u16) ?usize { for (&s.slots, 0..) |*sl, i| if (sl.used and sl.tag == tag) return i; return null; } fn setUname(s: *Self, uname: []const u8) void { const n = @min(uname.len, name_max); @memcpy(s.uname[0..n], uname[0..n]); s.uname_len = @intCast(n); } // -- the transport seam ---------------------------------------------- /// Offer parked requests back, one per call, in arrival order. Call in /// a loop until null, once per frame, BEFORE `next()`: the null both /// ends the round and resets it, so every parked request gets exactly /// one attempt per frame and a permanently blocked reader cannot /// starve the others. `src/fs_service.zig:196-208` is the contract and /// `src/fuse.zig:1166` is the other implementation of it. pub fn retry(s: *Self) ?fs.Req { var best: ?usize = null; for (&s.slots, 0..) |*sl, i| { if (!sl.used or !sl.parked or sl.retried) continue; if (best == null or sl.seq < s.slots[best.?].seq) best = i; } const i = best orelse { for (&s.slots) |*sl| sl.retried = false; return null; }; // No room for the answer is the end of the round too, and it must // reset it: leaving the flags set would make the next frame skip // the requests this one never reached. if (!s.hasRoom()) { for (&s.slots) |*sl| sl.retried = false; return null; } s.slots[i].retried = true; // In flight again: `reply` re-parks it if the core still has // nothing to say. s.slots[i].parked = false; return s.slots[i].req; } /// The next request off the wire, or null when there is nothing more to /// do with the bytes pushed so far. Call in a loop until null. /// /// ONE 9P MESSAGE IS NOT ONE REQUEST, which is the whole reason this is /// a state machine: a three-element `Twalk` is three lookups, a /// `Topen` with `OTRUNC` is a truncate and then an open, and a /// `Tversion` is none at all. So this pump decodes a message when it /// needs one, hands out its steps as the core answers them, and /// answers null only when the input is exhausted, the queue is full, or /// the core is holding a step. pub fn next(s: *Self) ?fs.Req { while (true) { if (s.job.kind != .none) { // A step is out with the core; the caller owes us a // `reply` before there is anything else to ask. if (s.job.req_tag != 0) return null; if (s.stepJob()) |req| return req; // The job answered itself — a walk that finished, an error // — and `stepJob` cleared it. Round again for the next // message. assert(s.job.kind == .none); continue; } if (s.orphan()) |req| return req; if (!s.hasRoom()) return null; if (!s.startFrame()) return null; } } /// Answer one request: queue the 9P reply it completes, advance the /// message it is a step of, or park it. `bytes` is the payload the /// core resolved and is borrowed for the duration of this call only — /// the same rule `src/fs_service.zig:224-229` states for the FUSE /// transport. pub fn reply(s: *Self, r: *const fs.Reply, bytes: []const u8) void { if (s.job.kind != .none and s.job.req_tag == r.tag) return s.jobReply(r, bytes); if (s.findSlot(r.tag)) |i| return s.slotReply(i, r, bytes); // An orphan's release, a park `Tversion` abandoned, or a request // `Tflush` already answered. Nothing to say and nobody to say it // to; `fuse.zig:1189` drops the same case for the same reason. } /// A `release` nobody is waiting for: the fid it belonged to is gone /// (the connection dropped, or `Tversion` reset it) but the core's /// open is not. /// /// The slot is freed HERE rather than when the answer lands, because /// nothing in the answer is wanted and `reply` already ignores a tag it /// no longer holds. That also means a release the core parks is /// dropped, which is the same trade `fuse.zig` makes for a write: a /// release is a transaction in this design and does not block. fn orphan(s: *Self) ?fs.Req { for (&s.fids) |*f| { if (!f.used or !f.orphan) continue; assert(f.open); const req: fs.Req = .{ .tag = s.tick(), .op = .release, .node = f.node, .handle = f.handle, }; f.* = .{}; return req; } return null; } /// Decode the message at the head of `in` and start serving it. False /// when there is not a whole one there yet. /// /// The frame stays in `in` for as long as the message is being served, /// because every string in a decoded `Msg` points into it. The `defer` /// is what makes that airtight: a message that answered itself here /// releases the frame immediately, and one that became a job hands the /// frame to the job, which releases it in `finishJob` or copies what it /// needs in `parkJob`. fn startFrame(s: *Self) bool { assert(s.job.kind == .none); assert(s.frame == 0); if (s.dead) return false; const len = frameLen(s.in[0..s.in_len]) orelse return false; // A `size` no encoder produced, or one this connection could never // buffer: either way the stream is not 9P and waiting for more of // it is waiting forever. if (len < header_len or len > s.in.len) { s.dead = true; return false; } if (len > s.in_len) return false; s.frame = len; defer if (s.job.kind == .none) s.dropFrame(); const got = decode(s.in[0..len]) catch { // The tag sits at a fixed offset and survives every way the // body can be wrong, so the client still gets an answer rather // than a hang. `len >= header_len` was checked above. s.fail(std.mem.readInt(u16, s.in[5..7], .little), e_botch); return true; }; // A server reads T-messages. An R-message here is a client on the // wrong end of the connection, or the double-role link // docs/9p.typ §7 tells us not to build. if (!isT(got.msg.msgType())) { s.fail(got.tag, e_botch); return true; } // «The client must communicate the version before any other // messages» — and until it has, there is no msize to bound // anything by. if (s.msize == 0 and got.msg != .tversion) { s.fail(got.tag, e_botch); return true; } if (s.msize != 0 and len > s.msize) { s.fail(got.tag, e_botch); return true; } s.dispatch(got); return true; } /// Release the served frame and slide the rest of the input down. The /// move is one message long and happens once per message; the /// alternative is a ring buffer, which would mean a decoded `Msg` /// could straddle the wrap and no longer be one slice. fn dropFrame(s: *Self) void { assert(s.frame != 0); assert(s.frame <= s.in_len); const n = s.frame; std.mem.copyForwards(u8, s.in[0 .. s.in_len - n], s.in[n..s.in_len]); s.in_len -= n; s.frame = 0; } // -- the messages ---------------------------------------------------- /// One T-message onto its handler. Every message either answers itself /// here or becomes `job`. fn dispatch(s: *Self, got: Decoded) void { switch (got.msg) { .tversion => |m| s.version(got.tag, m.msize, m.version), // REFUSED, all three, and each for its own reason. // // `Tauth`: there is no authentication here and there is not // going to be one in this file. The socket's permissions are // the protection and a network is tunnelled (docs/9p.typ §10). // // `Tcreate` and `Tremove`: the shape of this tree follows the // pane list, so there is nothing in it for a client to make or // unmake. The one place a client DOES create something is // `new/`, where walking to a name is what creates a pane // (`acmefs.zig:901-919`) — so the capability is there and it // is not spelled `Tcreate`. That answers the open question in // `docs/registry.typ` `9P-18`, and it takes most of `ad`'s // shipped-and-fixed bug list off the table with it. .tauth => s.fail(got.tag, e_no_auth), .tcreate => s.fail(got.tag, e_perm), .tattach => |m| s.attach(got.tag, m.fid, m.uname, m.aname), .tflush => |m| s.flush(got.tag, m.oldtag), .twalk => |m| s.walk(got, m.fid, m.newfid, @intCast(m.nwname)), .topen => |m| s.open(got.tag, m.fid, m.mode), .tread => |m| s.read(got.tag, m.fid, m.offset, m.count), .twrite => |m| s.write(got, m.fid, m.offset, m.data.len), .tclunk => |m| s.clunk(got.tag, m.fid, .clunk), // A remove clunks the fid too — see `clunk` — which is the // half of `remove(5)` that is easy to miss. .tremove => |m| s.clunk(got.tag, m.fid, .remove), .tstat => |m| s.stat(got.tag, m.fid), .twstat => |m| s.wstat(got.tag, m.fid, m.stat), // The R-variants, which `startFrame` already refused by // parity. Answered rather than `unreachable`, because the cost // of being wrong about that is a panic in a server. else => s.fail(got.tag, e_botch), } } /// `Tversion`: the msize handshake, and a connection reset. fn version(s: *Self, tag: u16, want: u32, ver: []const u8) void { // Three ceilings and the smallest wins: what the client will // accept, what one input buffer holds, and half of what the output // queue holds (`Options.out`). const cap: u32 = @intCast(@min(s.in.len, s.out.len / 2)); const m = @min(want, cap); if (m < msize_min) return s.fail(tag, e_small_msize); // u9fs `rversion`: any version string that STARTS with "9P" is // answered "9P2000", which is how a `.u` or `.L` client is told to // fall back to the base protocol. Anything else has no dialect in // common with us, and that is a SUCCESSFUL `Rversion` carrying the // literal "unknown" rather than an `Rerror`. const known = std.mem.startsWith(u8, ver, "9P"); s.reset(); // The msize only becomes real once a version is agreed: after // "unknown" the client must negotiate again, and `startFrame` // serves nothing else until it does. s.msize = if (known) m else 0; s.emit(tag, .{ .rversion = .{ .msize = m, .version = if (known) "9P2000" else "unknown" } }); } fn attach(s: *Self, tag: u16, fid: u32, uname: []const u8, aname: []const u8) void { // No `aname`. There is one tree here and it has no name; a client // that asked for another one is told so rather than handed this. if (aname.len != 0) return s.fail(tag, e_no_tree); if (fid == nofid) return s.fail(tag, e_unknown_fid); if (s.findFid(fid) != null) return s.fail(tag, e_fid_in_use); if (s.freeFid() == null) return s.fail(tag, e_too_many_fids); s.setUname(uname); // The root's attributes come from the core like every other node's. // Its node id is the only thing we were told (see `root`). s.job = .{ .kind = .attach, .tag = tag, .fid = fid, .node = s.root }; } fn walk(s: *Self, got: Decoded, fid: u32, newfid: u32, nwname: u8) void { const tag = got.tag; const i = s.findFid(fid) orelse return s.fail(tag, e_unknown_fid); // «must not have been opened for I/O» — walk(5). The fid IS the // open, so a walk would move the file out from under it. if (s.fids[i].open) return s.fail(tag, e_bad_use); if (newfid == nofid) return s.fail(tag, e_unknown_fid); if (newfid != fid) { if (s.findFid(newfid) != null) return s.fail(tag, e_fid_in_use); // Checked BEFORE any lookup, because a lookup under `new/` // creates a pane and a walk that then failed for want of a fid // slot would leave one behind. if (s.freeFid() == null) return s.fail(tag, e_too_many_fids); } if (nwname == 0) { // THE CLONE. No names, no lookups, no qids: `Rwalk` with // `nwqid == 0`, and it is a success — which is exactly why a // failure on the first element may not be spelled that way. if (newfid != fid) { const j = s.freeFid().?; s.fids[j] = s.fids[i]; s.fids[j].fid = newfid; // A clone shares the file and NOT the directory cursor: // two fids on one directory each keep their own place, // which is what a duplicated descriptor means everywhere // else. The open state is not shared either, and cannot // be — an open fid was refused above. s.fids[j].diroff = 0; s.fids[j].dirindex = 0; } s.emit(tag, .{ .rwalk = .{ .nwqid = 0 } }); return; } if (!s.fids[i].dir) return s.fail(tag, e_not_dir); s.job = .{ .kind = .walk, .tag = tag, .fid = fid, .newfid = newfid, .nwname = nwname, .node = s.fids[i].node, .dir = s.fids[i].dir, .perm = s.fids[i].perm, .name = s.fids[i].name, .name_len = s.fids[i].name_len, .msg = got.msg, }; } fn open(s: *Self, tag: u16, fid: u32, mode: u8) void { const i = s.findFid(fid) orelse return s.fail(tag, e_unknown_fid); const f = &s.fids[i]; if (f.open) return s.fail(tag, e_already_open); // Nothing in a generated tree can be removed, so nothing in it can // be opened remove-on-close either. if (mode & orclose != 0) return s.fail(tag, e_perm); const rw = mode & 3; if (rw == oexec) return s.fail(tag, e_perm); // A directory is read, and only read: 9P has no other verb for one, // and truncating a pane list is not a thing to mean. if (f.dir and (rw != oread or mode & otrunc != 0)) return s.fail(tag, e_perm); var need: u16 = 0; if (rw == oread or rw == ordwr) need |= 0o400; if (rw == owrite or rw == ordwr or mode & otrunc != 0) need |= 0o200; // THE PERMISSION CHECK IS OURS. Under FUSE the kernel does it, // against the mode a `getattr` reported, and `acmefs.open` never // sees a mode at all (`acmefs.zig:1023-1031`). Over 9P there is // nobody above us, so this is what stops `errors` and `wrsel` — // write-only in acme's own dirtab — from being readable. if (f.perm & need != need) return s.fail(tag, e_perm); s.job = .{ .kind = .open, .tag = tag, .fid = fid, .omode = mode }; } fn read(s: *Self, tag: u16, fid: u32, offset: u64, count: u32) void { const i = s.findFid(fid) orelse return s.fail(tag, e_unknown_fid); const f = &s.fids[i]; // The two conditions `u9fs.c:755-758` refuses, and the same // answer: a fid that was never opened, or one opened write-only. if (!f.open or (f.omode & 3) == owrite) return s.fail(tag, e_bad_use); // THE CLAMP, `min(count, msize - 11)`. An `Rread` longer than the // count asked for is a hard `-EIO` in Linux rather than a // truncation (`net/9p/client.c:1475-1479`, `9P-17`), and one // longer than the msize is a message the client cannot read at // all. Eleven is `Rread`'s header: `size[4] type[1] tag[2] // count[4]`. It is applied to the request as well as to the // answer, so the core is never asked to produce bytes that would // have to be thrown away. const want = @min(count, s.msize - header_len - 4); if (!f.dir) { s.job = .{ .kind = .read, .tag = tag, .fid = fid, .offset = offset, .count = want }; return; } // THE DIRECTORY RULE: offset zero, or exactly where the last read // ended, and nothing else (`u9fs.c:760-769`, `lib9p/srv.c:473`). A // client that seeks inside a directory is refused rather than // served a listing that tears — which is the bug `ad` has, where an // arbitrary offset that happens to land on an entry boundary is // silently accepted (`9P-5`). if (offset != f.diroff) { if (offset != 0) return s.fail(tag, e_bad_offset); f.diroff = 0; f.dirindex = 0; } s.job = .{ .kind = .readdir, .tag = tag, .fid = fid, .offset = offset, .count = want }; } fn write(s: *Self, got: Decoded, fid: u32, offset: u64, len: usize) void { const tag = got.tag; const i = s.findFid(fid) orelse return s.fail(tag, e_unknown_fid); const f = &s.fids[i]; if (!f.open or (f.omode & 3) == oread) return s.fail(tag, e_bad_use); s.job = .{ .kind = .write, .tag = tag, .fid = fid, .offset = offset, .count = @intCast(len), .msg = got.msg, }; } fn clunk(s: *Self, tag: u16, fid: u32, kind: Kind) void { assert(kind == .clunk or kind == .remove); const i = s.findFid(fid) orelse return s.fail(tag, e_unknown_fid); // An open fid owes the core a `release` before it goes. That is // what decrements a pane's `event` reader count, and losing it // leaves the editor reporting button actions to a script that has // gone (`acmefs.zig:1069-1093`). if (s.fids[i].open) { s.job = .{ .kind = kind, .tag = tag, .fid = fid }; return; } s.fids[i] = .{}; if (kind == .remove) s.fail(tag, e_perm) else s.emit(tag, .rclunk); } fn stat(s: *Self, tag: u16, fid: u32) void { if (s.findFid(fid) == null) return s.fail(tag, e_unknown_fid); s.job = .{ .kind = .stat, .tag = tag, .fid = fid }; } fn wstat(s: *Self, tag: u16, fid: u32, st: Stat) void { if (s.findFid(fid) == null) return s.fail(tag, e_unknown_fid); // The sentinels `stat(5)` specifies: an empty string and an // all-ones integer mean "do not touch". The codec above carries // them and has no opinion; deciding is this file's job. if (st.name.len != 0) return s.fail(tag, e_wstat); if (st.length == std.math.maxInt(u64)) { // Nothing left that we honour. Mode, owner, group and the two // times are ACCEPTED AND IGNORED, which is what a filesystem // of live editor state has to do with them // (`acmefs.zig:1096-1104`): refusing would make `touch` and // `chmod` fail on a tree where they mean nothing anyway. s.emit(tag, .rwstat); return; } // A length that is neither the sentinel nor zero. The core honours // exactly one value, so name it — and this string is one Linux // already knows, so `truncate` gets EPERM rather than 526. if (st.length != 0) return s.fail(tag, e_trunc_only); // ...and zero IS the truncate, which is the same `Req.truncate` // that `Topen` with `OTRUNC` produces (`FIX-1`). s.job = .{ .kind = .wstat, .tag = tag, .fid = fid }; } /// `Tflush`: a park-table lookup, and THE ORDER IS THE POINT. /// /// The original is answered first and the `Rflush` second. That is what /// `lib9p/srv.c:241-266` does with its chained flush list, and what /// `srv.c:810-827` does when the original finally responds: write the /// original's reply, then respond to every flush waiting on it. A /// client that sees `Rflush` may reuse the tag, so a reply arriving /// after it would be a reply to whatever the tag names NEXT. /// /// `docs/registry.typ` `9P-16` says this is where we beat the prior /// art, and the reason is structural rather than clever: the park table /// is already keyed per outstanding request, so this is a lookup and /// two replies. `ad` gets the ordering right in thirty-nine lines and /// then defaults its filesystem's `flush` hook to doing nothing, so a /// client flushing a blocked `event` read waits for an unrelated editor /// event to arrive. There is no hook here to forget to implement. fn flush(s: *Self, tag: u16, oldtag: u16) void { if (s.findTag(oldtag)) |i| { // EINTR and drop it, which is exactly what `fuse.zig:1334-1341` // answers a `FUSE_INTERRUPT` naming a parked request. s.fail(s.slots[i].tag, e_interrupted); s.slots[i] = .{}; } // A tag we do not hold was already answered or never existed. // `Rflush` either way: after it the client may reuse the tag, and // that is the only promise `flush(5)` makes. s.emit(tag, .rflush); } // -- steps and answers ----------------------------------------------- /// Record the step being handed to the core, so that a park has /// something to copy and `reply` has something to match. fn ask(s: *Self, req: fs.Req) fs.Req { assert(req.tag != 0); s.job.req = req; s.job.req_tag = req.tag; return req; } /// The fid the message in flight names. Null cannot happen — nothing /// else runs while a job does — and is answered rather than asserted, /// because the cost of being wrong is a corrupted table. fn jobFid(s: *Self) ?*Fid { const i = s.findFid(s.job.fid) orelse { s.fail(s.job.tag, e_unknown_fid); s.finishJob(); return null; }; return &s.fids[i]; } /// The next core request the message in flight needs, or null when it /// has just answered itself. fn stepJob(s: *Self) ?fs.Req { const j = &s.job; assert(j.kind != .none); assert(j.req_tag == 0); switch (j.kind) { .none => unreachable, .attach => return s.ask(.{ .tag = s.tick(), .op = .getattr, .node = s.root }), .walk => return s.stepWalk(), .open => { const f = s.jobFid() orelse return null; // `OTRUNC` is a truncate and THEN an open, in that order. if (j.step == 0 and j.omode & otrunc != 0) return s.ask(.{ .tag = s.tick(), .op = .setattr, .node = f.node, .truncate = true, }); return s.ask(.{ .tag = s.tick(), .op = .open, .node = f.node }); }, .read => { const f = s.jobFid() orelse return null; return s.ask(.{ .tag = s.tick(), .op = .read, .node = f.node, .handle = f.handle, .off = j.offset, .size = j.count, }); }, .readdir => { const f = s.jobFid() orelse return null; // THE COORDINATE CHANGE. 9P counts bytes and `acmefs` // counts entries (`acmefs.zig:972`), so the request carries // the entry index this fid's byte cursor stands at, and // `emitDirRead` advances both. return s.ask(.{ .tag = s.tick(), .op = .readdir, .node = f.node, .handle = f.handle, .off = f.dirindex, .size = j.count, }); }, .write => { const f = s.jobFid() orelse return null; return s.ask(.{ .tag = s.tick(), .op = .write, .node = f.node, .handle = f.handle, .off = j.offset, .size = j.count, .data = j.msg.twrite.data, }); }, .clunk, .remove => { const f = s.jobFid() orelse return null; return s.ask(.{ .tag = s.tick(), .op = .release, .node = f.node, .handle = f.handle, }); }, .stat => { const f = s.jobFid() orelse return null; return s.ask(.{ .tag = s.tick(), .op = .getattr, .node = f.node }); }, .wstat => { const f = s.jobFid() orelse return null; return s.ask(.{ .tag = s.tick(), .op = .setattr, .node = f.node, .truncate = true }); }, } } /// One walk element at a time, and the local ones without asking. fn stepWalk(s: *Self) ?fs.Req { const j = &s.job; while (j.step < j.nwname) { const name = j.msg.twalk.wname[j.step]; // A name the fid could not hold cannot be a name in this tree, // and refusing it on its length is what keeps `Rstat` honest. if (name.len > name_max) { s.stopWalk(e_illegal_name); return null; } // `.` is the fid where it already stands, and costs nothing. if (std.mem.eql(u8, name, ".")) { j.wqid[j.nwqid] = qidOf(j.node, j.dir); j.nwqid += 1; j.step += 1; continue; } if (std.mem.eql(u8, name, "..")) { const p = parentOf(j.node, s.root, &j.name); j.name_len = p.name_len; j.node = p.node; // WHICH node the parent is, is ours to work out; what it // LOOKS like is not. A `getattr` keeps `perm`, `dir` and // the qid the core's answer rather than this file's // invention, and reports ENOENT if the pane closed // underneath us. return s.ask(.{ .tag = s.tick(), .op = .getattr, .node = p.node }); } @memcpy(j.name[0..name.len], name); j.name_len = @intCast(name.len); return s.ask(.{ .tag = s.tick(), .op = .lookup, .node = j.node, .data = name }); } // Every element resolved, so `newfid` is bound — and only now. A // partial walk binds NOTHING, which is `ad`'s «new_fid is only // bound when all elements were walked successfully» and the spec's. const dst = pick: { if (j.newfid == j.fid) break :pick s.findFid(j.fid) orelse { s.fail(j.tag, e_unknown_fid); s.finishJob(); return null; }; break :pick s.freeFid() orelse { s.fail(j.tag, e_too_many_fids); s.finishJob(); return null; }; }; s.fids[dst] = .{ .used = true, .fid = j.newfid, .node = j.node, .dir = j.dir, .perm = j.perm, .name = j.name, .name_len = j.name_len, }; s.emit(j.tag, .{ .rwalk = .{ .nwqid = j.nwqid, .wqid = j.wqid } }); s.finishJob(); return null; } /// A walk that could not finish, and THE SCAR that says how to answer /// it: «Spec: first element failure must be Rerror, not Rwalk with zero /// qids» — `ad/crates/ninep/src/sansio/server.rs:335-338`, left in /// their source after they shipped it the other way. Zero qids already /// means the clone, so it cannot also mean a failure. /// /// A failure at any LATER element is a successful short `Rwalk`, and /// the client is expected to notice that it got fewer qids than it /// asked for. It gets no error string at all, which is the protocol's /// choice and not ours. fn stopWalk(s: *Self, ename: []const u8) void { const j = &s.job; if (j.nwqid == 0) s.fail(j.tag, ename) else s.emit(j.tag, .{ .rwalk = .{ .nwqid = j.nwqid, .wqid = j.wqid } }); s.finishJob(); } fn finishJob(s: *Self) void { s.job = .{}; if (s.frame != 0) s.dropFrame(); } /// The core answered a step of the message in flight. fn jobReply(s: *Self, r: *const fs.Reply, bytes: []const u8) void { const j = &s.job; assert(j.kind != .none); assert(j.req_tag == r.tag); j.req_tag = 0; // A CLUNK CANNOT FAIL. «even if the clunk fails, the fid is no // longer valid» — clunk(5) — and `remove(5)` says the same of // remove, so the core's answer to the release is not consulted at // all. That also means a release the core tried to park is dropped // rather than leaving behind a fid the client can no longer reach. if (j.kind == .clunk or j.kind == .remove) { s.dropFid(j.fid); if (j.kind == .remove) s.fail(j.tag, e_perm) else s.emit(j.tag, .rclunk); s.finishJob(); return; } if (r.status == .again) return s.parkJob(); if (r.status == .err) { const ename = errString(r.errno); if (j.kind == .walk) return s.stopWalk(ename); s.fail(j.tag, ename); s.finishJob(); return; } switch (j.kind) { .none, .clunk, .remove => unreachable, .attach => { const i = s.freeFid() orelse { s.fail(j.tag, e_too_many_fids); s.finishJob(); return; }; const node = if (r.attr.node != 0) r.attr.node else s.root; s.fids[i] = .{ .used = true, .fid = j.fid, .node = node, .dir = r.attr.dir, .perm = r.attr.mode, }; // The root's name is "/" — one of the bugs `ad` shipped // and then fixed (`9P-18`, commit `64f2f4b`). s.fids[i].name[0] = '/'; s.fids[i].name_len = 1; s.emit(j.tag, .{ .rattach = .{ .qid = qidOf(node, r.attr.dir) } }); s.finishJob(); }, .walk => { if (r.attr.node != 0) j.node = r.attr.node; j.dir = r.attr.dir; j.perm = r.attr.mode; j.wqid[j.nwqid] = qidOf(j.node, j.dir); j.nwqid += 1; j.step += 1; // The job STAYS: `next()` asks `stepWalk` for the next // element, or lets it bind the fid and answer. }, .open => { if (j.step == 0 and j.omode & otrunc != 0) { // The truncate landed; the open is the next step. j.step = 1; return; } const f = s.jobFid() orelse return; f.open = true; f.omode = j.omode; f.handle = r.handle; // A fresh open starts a directory at the beginning. f.diroff = 0; f.dirindex = 0; // `iounit` is the largest atomic read or write: one message // less the slack `fcall.h:72` has reserved for a `Twrite` // header for thirty years. s.emit(j.tag, .{ .ropen = .{ .qid = qidOf(f.node, f.dir), .iounit = s.msize - iohdrsz, } }); s.finishJob(); }, .read => { // Clamped a second time, against the bytes that actually // came back: the request already carried the count, and a // core that answered with more would otherwise become an // `-EIO` in the client rather than a bug here. s.emit(j.tag, .{ .rread = .{ .data = bytes[0..@min(bytes.len, j.count)] } }); s.finishJob(); }, .readdir => { s.emitDirRead(j.tag, j.fid, bytes, j.count); s.finishJob(); }, .write => { // The core's own count and not the request's: `data` // refusing a partial grapheme is a real short write, and // claiming the whole request would tell the writer that // its trailing bytes landed when they did not. s.emit(j.tag, .{ .rwrite = .{ .count = @min(r.written, j.count) } }); s.finishJob(); }, .stat => { const f = s.jobFid() orelse return; // The core is authoritative about size and mode, and the // fid's cache follows: `body` grows between stats, and // `Topen` is checked against `perm`. f.perm = r.attr.mode; f.dir = r.attr.dir; s.emit(j.tag, .{ .rstat = .{ .stat = s.statOf(f, r.attr) } }); s.finishJob(); }, .wstat => { s.emit(j.tag, .rwstat); s.finishJob(); }, } } /// The core said `.again`: nothing consumed, ask me later. The request /// moves into a park slot and the 9P tag goes with it, so the client /// hears nothing at all until the core has something to say — which is /// what makes a blocking `event` read work on a single-threaded core /// with no waiter list anywhere. fn parkJob(s: *Self) void { const j = &s.job; // Only these three can park, and only because the state a retry // needs is scalars. A walk cannot: its names borrow the input // buffer, which the next message overwrites. `e_again` is honest // (the client may retry) and by construction unreachable — the // core parks reads of `event` and nothing else. switch (j.kind) { .read, .readdir, .write => {}, else => { s.fail(j.tag, e_again); s.finishJob(); return; }, } const i = s.freeSlot() orelse { // Overflow is a refusal, not a queue: thirty-two blocked // readers is thirty-two scripts watching one session. s.fail(j.tag, e_again); s.finishJob(); return; }; const sl = &s.slots[i]; sl.* = .{ .used = true, .parked = true, .seq = s.tick(), .tag = j.tag, .kind = j.kind, .fid = j.fid, .count = j.count, .req = j.req, }; if (j.req.data.len != 0) { if (j.req.data.len > park_data_max) { // A payload too large to copy would go on borrowing the // input buffer, so parking it would park a dangling slice. sl.* = .{}; s.fail(j.tag, e_again); s.finishJob(); return; } @memcpy(sl.data[0..j.req.data.len], j.req.data); sl.copied = true; sl.req.data = sl.data[0..j.req.data.len]; } // The frame is nobody's now: everything the retry needs has been // copied, so the next message may take its place. s.finishJob(); } /// The core answered a request that had been parked. fn slotReply(s: *Self, i: usize, r: *const fs.Reply, bytes: []const u8) void { const sl = &s.slots[i]; if (r.status == .again) { sl.parked = true; return; } if (r.status == .err) { s.fail(sl.tag, errString(r.errno)); sl.* = .{}; return; } switch (sl.kind) { .read => s.emit(sl.tag, .{ .rread = .{ .data = bytes[0..@min(bytes.len, sl.count)] } }), .readdir => s.emitDirRead(sl.tag, sl.fid, bytes, sl.count), .write => s.emit(sl.tag, .{ .rwrite = .{ .count = @min(r.written, sl.count) } }), // `parkJob` admits no other kind. else => s.fail(sl.tag, e_botch), } sl.* = .{}; } /// A directory read: `acmefs`'s staged entries become 9P `stat` /// records, in place, and the fid's cursor advances by exactly what /// was sent. /// /// THE ONE ENTRY THAT DID NOT FIT needs no buffer here, and that is /// worth saying because every reference server has one: u9fs caches a /// `dirent` per fid «for when convD2M fails» (`u9fs.c:780`) because /// `readdir(3)` has already consumed it. `acmefs` re-stages the whole /// listing from an entry index on every call and says why — /// «re-staging from scratch on every call is what makes a partially /// consumed answer safe to ask for again at a higher cookie» /// (`acmefs.zig:1013-1016`) — so an entry that does not fit is simply /// not counted, and the next read asks for it by index. fn emitDirRead(s: *Self, tag: u16, fid: u32, staging: []const u8, count: u32) void { const buf = s.out[s.out_len..]; assert(buf.len > header_len + 4); const cap = @min(@as(usize, count), buf.len - header_len - 4); const who = s.uname[0..s.uname_len]; var n: usize = header_len + 4; var entries: u32 = 0; var i: usize = 0; // `node[8] dir[1] namelen[1] name[]`, repeated — `acmefs.zig:942`. while (i + 10 <= staging.len) { const nlen = staging[i + 9]; if (i + 10 + nlen > staging.len) break; const dir = staging[i + 8] != 0; const rec: Stat = .{ .type = 0, .dev = 0, .qid = qidOf(std.mem.readInt(u64, staging[i..][0..8], .little), dir), .mode = (if (dir) dmdir else 0) | @as(u32, if (dir) dirent_dir_perm else dirent_file_perm), .atime = 0, .mtime = 0, .length = 0, .name = staging[i + 10 ..][0..nlen], .uid = who, .gid = who, .muid = who, }; const size = @as(usize, rec.size() catch break) + 2; // WHOLE RECORDS ONLY. `read(5)`: a directory read returns an // integral number of entries, so the first one that does not // fit ends the reply and the cursor stops in front of it. if (n - header_len - 4 + size > cap) break; _ = rec.encode(buf[n..]) catch break; n += size; entries += 1; i += 10 + nlen; } // A count that cannot hold the FIRST entry is refused rather than // answered with zero bytes, because zero bytes is end of directory // and a client that believes it stops asking. `entries == 0` with // nothing staged is the real end. if (entries == 0 and staging.len != 0) return s.fail(tag, e_count_small); const payload: u32 = @intCast(n - header_len - 4); // The header goes on LAST, over bytes reserved for it, because the // payload's length is only known once the entries are encoded — // `u9fs.c:775-806` builds it the same way and for the same reason. // Written by hand rather than through `encode`, which would want // the payload contiguous somewhere else first, and this file will // not carry a third msize buffer to make that true. The test "a // directory read is whole stat records" decodes the result with // `decode`, which is what keeps these four lines honest. comptime assert(header_len == 7); std.mem.writeInt(u32, buf[0..4], @intCast(n), .little); buf[4] = @intFromEnum(Type.rread); std.mem.writeInt(u16, buf[5..7], tag, .little); std.mem.writeInt(u32, buf[7..11], payload, .little); s.out_len += n; // BOTH cursors, together, or the next read is refused: bytes for // the client's offset rule, entries for `acmefs`'s index. if (s.findFid(fid)) |k| { s.fids[k].diroff += payload; s.fids[k].dirindex += entries; } } /// One `stat` record for a file the core has just described. /// /// `type` and `dev` are Plan 9 kernel device identifiers, meaningless /// off Plan 9, and zero — as u9fs sends them. `atime` and `mtime` are /// zero because this tree has no times to report and the FUSE /// transport already reports none (`fuse.zig:1544-1546`); an invented /// time is one `make` would believe. The three name fields are whoever /// attached: the tree is synthetic and has exactly one owner. fn statOf(s: *const Self, f: *const Fid, a: fs.Reply.Attr) Stat { const who = s.uname[0..s.uname_len]; return .{ .type = 0, .dev = 0, .qid = qidOf(if (a.node != 0) a.node else f.node, a.dir), // The high bits are the type and the low nine are the // permission: `dmdir` is `qtdir` shifted up 24, which the // codec's last test asserts rather than assumes. .mode = (if (a.dir) dmdir else 0) | @as(u32, a.mode), .atime = 0, .mtime = 0, .length = a.size, .name = f.name[0..f.name_len], .uid = who, .gid = who, .muid = who, }; } }; } // --------------------------------------------------------------------------- // server tests // --------------------------------------------------------------------------- // // Driven with BYTE ARRAYS and a STUB FILESYSTEM, so there is no `Pardes` here // and no transport either: `push` takes encoded messages, `pump` is // `fs_service.drain` written out, and `reap` decodes what came back with the // codec above. A test that fails is a message a real client would have been // sent, byte for byte. /// Everything `Server` asks of a filesystem, plus a tree small enough to check /// by eye. The three types are `acmefs`'s ABI verbatim — that is the whole /// contract, and `Server(acmefs)` is the instantiation that matters — so this /// is a MIRROR and not a redefinition: a field that drifts is a compile error /// the moment the real adapter is built. /// /// THE NODE IDS ARE `acmefs.Node`'s PACKING, `{ file: u4, serial: u60 }`, and /// they have to be: `parentOf` reads them. So pane 1's directory is `1 << 4`, /// its `body` is that plus `PaneFile.body` (2), and the top-level files are /// `TopFile`'s own 1..4 with a zero serial. const StubFs = struct { pub const Op = enum(u8) { lookup, getattr, setattr, open, read, write, release, readdir, statfs }; pub const Status = enum(u8) { ok, again, err }; pub const Req = struct { tag: u64, op: Op, node: u64, handle: u32 = 0, off: u64 = 0, size: u32 = 0, data: []const u8 = &.{}, truncate: bool = false, }; pub const Reply = struct { tag: u64, status: Status = .ok, errno: u16 = 0, attr: Attr = .{}, handle: u32 = 0, written: u32 = 0, pub const Attr = struct { node: u64 = 0, dir: bool = false, size: u64 = 0, mode: u16 = 0o600, }; }; const Entry = struct { node: u64, parent: u64, name: []const u8, dir: bool, mode: u16 }; /// `acmefs`'s tree, cut down: the root's four names, two panes, and six of /// a pane's files including the two that matter most here — `event`, which /// blocks, and `errors`, which is write-only. const tree = [_]Entry{ .{ .node = 1, .parent = 1, .name = "/", .dir = true, .mode = 0o500 }, .{ .node = 2, .parent = 1, .name = "index", .dir = false, .mode = 0o400 }, .{ .node = 3, .parent = 1, .name = "cons", .dir = false, .mode = 0o200 }, .{ .node = 4, .parent = 1, .name = "new", .dir = true, .mode = 0o500 }, .{ .node = 16, .parent = 1, .name = "1", .dir = true, .mode = 0o500 }, .{ .node = 32, .parent = 1, .name = "2", .dir = true, .mode = 0o500 }, .{ .node = 17, .parent = 16, .name = "addr", .dir = false, .mode = 0o600 }, .{ .node = 18, .parent = 16, .name = "body", .dir = false, .mode = 0o600 }, .{ .node = 19, .parent = 16, .name = "ctl", .dir = false, .mode = 0o600 }, .{ .node = 21, .parent = 16, .name = "errors", .dir = false, .mode = 0o200 }, .{ .node = 22, .parent = 16, .name = "event", .dir = false, .mode = 0o600 }, .{ .node = 23, .parent = 16, .name = "tag", .dir = false, .mode = 0o600 }, }; const body_node = 18; const index_node = 2; const event_node = 22; body: []const u8 = "hello, body\n", /// `index`, and long enough that a read of it has to be clamped. filler: [1024]u8 = @splat('x'), /// One event record, or nothing — which is `Status.again`. event: ?[]const u8 = null, /// Set to make every write park, so the copy into a slot is exercised. park_writes: bool = false, releases: u32 = 0, calls: u32 = 0, writes: [128]u8 = undefined, writes_len: usize = 0, stage: [1024]u8 = undefined, const Answer = struct { reply: Reply, bytes: []const u8 = "" }; fn find(node: u64) ?usize { for (tree, 0..) |e, i| if (e.node == node) return i; return null; } fn sizeOf(st: *const StubFs, node: u64) u64 { return switch (node) { body_node => st.body.len, index_node => st.filler.len, else => 0, }; } fn contentOf(st: *const StubFs, node: u64) []const u8 { return switch (node) { body_node => st.body, index_node => &st.filler, else => "", }; } fn attrOf(st: *const StubFs, e: Entry) Reply.Attr { return .{ .node = e.node, .dir = e.dir, .mode = e.mode, .size = st.sizeOf(e.node) }; } /// `acmefs`'s staging format, which is what the 9P server decodes: /// `node[8] dir[1] namelen[1] name[]`, repeated, from an ENTRY INDEX. fn stageDir(st: *StubFs, node: u64, skip: u64) []const u8 { var n: usize = 0; var seen: u64 = 0; for (tree) |e| { if (e.parent != node or e.node == node) continue; if (seen < skip) { seen += 1; continue; } std.mem.writeInt(u64, st.stage[n..][0..8], e.node, .little); st.stage[n + 8] = @intFromBool(e.dir); st.stage[n + 9] = @intCast(e.name.len); @memcpy(st.stage[n + 10 ..][0..e.name.len], e.name); n += 10 + e.name.len; } return st.stage[0..n]; } fn handle(st: *StubFs, req: Req) Answer { st.calls += 1; const fail: Answer = .{ .reply = .{ .tag = req.tag, .status = .err, .errno = 2 } }; const i = find(req.node) orelse return fail; switch (req.op) { .lookup => { for (tree) |e| { if (e.parent != req.node or e.node == req.node) continue; if (!std.mem.eql(u8, e.name, req.data)) continue; return .{ .reply = .{ .tag = req.tag, .attr = st.attrOf(e) } }; } return fail; }, .getattr => return .{ .reply = .{ .tag = req.tag, .attr = st.attrOf(tree[i]) } }, .setattr => { if (req.truncate and req.node == body_node) st.body = ""; return .{ .reply = .{ .tag = req.tag, .attr = st.attrOf(tree[i]) } }; }, .open => return .{ .reply = .{ .tag = req.tag, .handle = 7 } }, .release => { st.releases += 1; return .{ .reply = .{ .tag = req.tag } }; }, .readdir => { if (!tree[i].dir) return .{ .reply = .{ .tag = req.tag, .status = .err, .errno = 20 } }; return .{ .reply = .{ .tag = req.tag }, .bytes = st.stageDir(req.node, req.off) }; }, .read => { // `event`: one record per read, and `.again` when there is // none — `acmefs.zig:1358-1367` exactly. if (req.node == event_node) { const rec = st.event orelse return .{ .reply = .{ .tag = req.tag, .status = .again } }; st.event = null; return .{ .reply = .{ .tag = req.tag }, .bytes = rec }; } const all = st.contentOf(req.node); if (req.off >= all.len) return .{ .reply = .{ .tag = req.tag } }; const from = all[@intCast(req.off)..]; return .{ .reply = .{ .tag = req.tag }, .bytes = from[0..@min(from.len, req.size)] }; }, .write => { if (st.park_writes) return .{ .reply = .{ .tag = req.tag, .status = .again } }; const n = @min(req.data.len, st.writes.len - st.writes_len); @memcpy(st.writes[st.writes_len..][0..n], req.data[0..n]); st.writes_len += n; return .{ .reply = .{ .tag = req.tag, .written = @intCast(n) } }; }, .statfs => return .{ .reply = .{ .tag = req.tag } }, } } }; const Srv = Server(StubFs); /// One connection: two buffers, a stub filesystem and the server between them. const Harness = struct { in: [4096]u8 = undefined, out: [8192]u8 = undefined, fsys: StubFs = .{}, srv: Srv = undefined, /// The buffers are fields, so the server can only be built once the /// harness has an address. fn start(h: *Harness) void { h.srv = Srv.init(.{ .in = &h.in, .out = &h.out, .root = 1 }); } fn answer(h: *Harness, req: StubFs.Req) void { const a = h.fsys.handle(req); h.srv.reply(&a.reply, a.bytes); } /// THE TRANSPORT CONTRACT, in the shape `src/fs_service.zig:209-222` /// requires it: every parked request offered once, then everything the /// wire has, both loops to null. Written out rather than imported — /// `fs_service` is `pardes.zig` deep — and writing it out is how these /// tests document what they are testing against. fn pump(h: *Harness) void { while (h.srv.retry()) |req| h.answer(req); while (h.srv.next()) |req| h.answer(req); } fn send(h: *Harness, tag: u16, msg: Msg) !void { var buf: [1024]u8 = undefined; const bytes = try encode(msg, tag, &buf); try testing.expectEqual(bytes.len, h.srv.push(bytes)); h.pump(); } /// One reply off the queue. Borrows the out buffer, so a caller checks it /// before sending anything else. fn reap(h: *Harness) !Decoded { const out = h.srv.output(); const len = frameLen(out) orelse return error.NoReply; if (len > out.len) return error.ShortReply; const got = try decode(out[0..len]); h.srv.wrote(len); return got; } fn quiet(h: *Harness) !void { try testing.expectEqual(@as(usize, 0), h.srv.output().len); } /// `Tversion` and `Tattach`, which every test but the handshake ones want, /// leaving the root on fid 0. fn handshake(h: *Harness, msize: u32) !void { h.start(); try h.send(notag, .{ .tversion = .{ .msize = msize, .version = "9P2000" } }); const v = try h.reap(); try testing.expectEqualStrings("9P2000", v.msg.rversion.version); try h.send(0, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "goblin", .aname = "" } }); const a = try h.reap(); try testing.expectEqual(@as(u64, 1), a.msg.rattach.qid.path); } /// A walk from the root to one name, landing on `newfid`. fn walkTo(h: *Harness, tag: u16, newfid: u32, list: []const []const u8) !Decoded { try h.send(tag, .{ .twalk = .{ .fid = 0, .newfid = newfid, .nwname = @intCast(list.len), .wname = wnames(list), } }); return h.reap(); } }; fn wnames(list: []const []const u8) [max_welem][]const u8 { var out: [max_welem][]const u8 = @splat(""); for (list, 0..) |n, i| out[i] = n; return out; } /// The names in a directory read, decoded as whole `stat` records. Which is /// also the proof that `emitDirRead`'s hand-written header is right: this goes /// through `Stat.decode`, which refuses anything that does not add up. fn dirNames(data: []const u8, out: [][]const u8) !usize { var n: usize = 0; var i: usize = 0; while (i < data.len) { const size = std.mem.readInt(u16, data[i..][0..2], .little); const st = try Stat.decode(data[i..][0 .. @as(usize, size) + 2]); out[n] = st.name; n += 1; i += @as(usize, size) + 2; } return n; } test "9p server: the version handshake clamps, falls back, and refuses" { var h: Harness = .{}; h.start(); // A client offering a megabyte gets what the buffers hold: one input // buffer, or half the output queue, whichever is smaller. try h.send(notag, .{ .tversion = .{ .msize = 1 << 20, .version = "9P2000" } }); var got = try h.reap(); try testing.expectEqual(notag, got.tag); try testing.expectEqual(@as(u32, 4096), got.msg.rversion.msize); try testing.expectEqualStrings("9P2000", got.msg.rversion.version); // BELOW LINUX'S FLOOR IS FINE. 4096 is the kernel's number and nobody // else's: Plan 9's devmnt, plan9port's `9p` and our own client all accept // 512, and refusing it would cost the board a kilobyte for nothing. try h.send(notag, .{ .tversion = .{ .msize = 512, .version = "9P2000" } }); got = try h.reap(); try testing.expectEqual(@as(u32, 512), got.msg.rversion.msize); // A `.u` client is told to fall back rather than refused: u9fs answers // "9P2000" to any version starting with "9P". try h.send(notag, .{ .tversion = .{ .msize = 4096, .version = "9P2000.u" } }); got = try h.reap(); try testing.expectEqualStrings("9P2000", got.msg.rversion.version); // Something that is not 9P at all: the literal "unknown", in a SUCCESSFUL // Rversion and not an Rerror. try h.send(notag, .{ .tversion = .{ .msize = 4096, .version = "TCP/IP" } }); got = try h.reap(); try testing.expectEqualStrings("unknown", got.msg.rversion.version); // ...and nothing else is served until a version is agreed. try h.send(1, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "goblin", .aname = "" } }); got = try h.reap(); try testing.expectEqualStrings(e_botch, got.msg.rerror.ename); // An msize too small to hold one `Rwalk` cannot be served at all, and // `Rversion` has no field that means "too small" — so `Rerror`, which is a // legal reply to any T-message. try h.send(notag, .{ .tversion = .{ .msize = 64, .version = "9P2000" } }); got = try h.reap(); try testing.expectEqualStrings(e_small_msize, got.msg.rerror.ename); try testing.expectEqual(@as(u32, 216), msize_min - 1); } test "9p server: attach names the root, and the only tree there is" { var h: Harness = .{}; h.start(); try h.send(notag, .{ .tversion = .{ .msize = 4096, .version = "9P2000" } }); _ = try h.reap(); // An `aname` names a tree we do not have, and saying so beats handing over // the one we do. try h.send(1, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "goblin", .aname = "work" } }); var got = try h.reap(); try testing.expectEqualStrings(e_no_tree, got.msg.rerror.ename); try h.send(2, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "goblin", .aname = "" } }); got = try h.reap(); try testing.expectEqual(@as(u64, 1), got.msg.rattach.qid.path); try testing.expectEqual(qtdir, got.msg.rattach.qid.type); // ALWAYS ZERO, which is what makes Linux's client skip its cache. try testing.expectEqual(@as(u32, 0), got.msg.rattach.qid.version); // The same fid twice is a client bug, refused rather than rebound. try h.send(3, .{ .tattach = .{ .fid = 0, .afid = nofid, .uname = "goblin", .aname = "" } }); got = try h.reap(); try testing.expectEqualStrings(e_fid_in_use, got.msg.rerror.ename); // `Tauth` is refused, which is how a Plan 9 mount learns there is no // authentication here and carries on without it. try h.send(4, .{ .tauth = .{ .afid = 1, .uname = "goblin", .aname = "" } }); got = try h.reap(); try testing.expectEqualStrings(e_no_auth, got.msg.rerror.ename); // The attacher's name is what `Rstat` reports as owner, group and last // modifier: the tree is synthetic and has exactly one owner. try h.send(5, .{ .tstat = .{ .fid = 0 } }); got = try h.reap(); try testing.expectEqualStrings("/", got.msg.rstat.stat.name); try testing.expectEqualStrings("goblin", got.msg.rstat.stat.uid); try testing.expectEqualStrings("goblin", got.msg.rstat.stat.muid); try testing.expect(got.msg.rstat.stat.mode & dmdir != 0); } test "9p server: a three-element walk, and `..` with no kernel to resolve it" { var h: Harness = .{}; try h.handshake(4096); // Three elements in one message, on a tree that is three deep: `..` at the // root is the root, which is POSIX's rule and intro(5)'s, and NOT an error. var got = try h.walkTo(5, 1, &.{ "..", "1", "body" }); try testing.expectEqual(@as(u16, 3), got.msg.rwalk.nwqid); try testing.expectEqual(@as(u64, 1), got.msg.rwalk.wqid[0].path); try testing.expectEqual(@as(u64, 16), got.msg.rwalk.wqid[1].path); try testing.expectEqual(@as(u64, 18), got.msg.rwalk.wqid[2].path); try testing.expectEqual(qtdir, got.msg.rwalk.wqid[1].type); try testing.expectEqual(qtfile, got.msg.rwalk.wqid[2].type); for (got.msg.rwalk.wqid[0..3]) |q| try testing.expectEqual(@as(u32, 0), q.version); // Up and down and up and down. `..` from a pane's directory is the root // too, and five elements land where two would have. got = try h.walkTo(6, 2, &.{ "..", "1", "..", "1", "body" }); try testing.expectEqual(@as(u16, 5), got.msg.rwalk.nwqid); try testing.expectEqual(@as(u64, 1), got.msg.rwalk.wqid[2].path); try testing.expectEqual(@as(u64, 18), got.msg.rwalk.wqid[4].path); // ...and the fid really is the body. try h.send(7, .{ .tstat = .{ .fid = 2 } }); got = try h.reap(); try testing.expectEqualStrings("body", got.msg.rstat.stat.name); try testing.expectEqual(@as(u64, 12), got.msg.rstat.stat.length); // `..` after an element that landed on a FILE is that pane's directory — // `parentOf`'s third case — and the name comes out of the node id rather // than out of the element, because "1" is not what the client typed. got = try h.walkTo(8, 3, &.{ "1", "body", ".." }); try testing.expectEqual(@as(u16, 3), got.msg.rwalk.nwqid); try testing.expectEqual(@as(u64, 16), got.msg.rwalk.wqid[2].path); try testing.expectEqual(qtdir, got.msg.rwalk.wqid[2].type); try h.send(9, .{ .tstat = .{ .fid = 3 } }); got = try h.reap(); try testing.expectEqualStrings("1", got.msg.rstat.stat.name); // `.` is where the fid already stands, and it costs the core NOTHING: no // request is made for it at all. const before = h.fsys.calls; got = try h.walkTo(10, 4, &.{ ".", "." }); try testing.expectEqual(@as(u16, 2), got.msg.rwalk.nwqid); try testing.expectEqual(@as(u64, 1), got.msg.rwalk.wqid[1].path); try testing.expectEqual(before, h.fsys.calls); } test "9p server: a walk failing on the first element is Rerror, on the second a short Rwalk" { var h: Harness = .{}; try h.handshake(4096); // THE SCAR (`ad/.../sansio/server.rs:335-338`): zero qids already means // the clone, so a failure on the first element cannot be spelled that way. var got = try h.walkTo(5, 1, &.{ "nope", "body" }); try testing.expectEqualStrings("No such file or directory", got.msg.rerror.ename); // A failure LATER is a successful short `Rwalk` with no error string at // all, which is the protocol's choice and not ours. got = try h.walkTo(6, 1, &.{ "1", "nope" }); try testing.expectEqual(@as(u16, 1), got.msg.rwalk.nwqid); try testing.expectEqual(@as(u64, 16), got.msg.rwalk.wqid[0].path); // ...and `newfid` is NOT bound by a partial walk. try h.send(7, .{ .tstat = .{ .fid = 1 } }); got = try h.reap(); try testing.expectEqualStrings(e_unknown_fid, got.msg.rerror.ename); // The clone does bind it, with zero qids and no lookups. try h.send(8, .{ .twalk = .{ .fid = 0, .newfid = 1, .nwname = 0 } }); got = try h.reap(); try testing.expectEqual(@as(u16, 0), got.msg.rwalk.nwqid); try h.send(9, .{ .tstat = .{ .fid = 1 } }); got = try h.reap(); try testing.expectEqualStrings("/", got.msg.rstat.stat.name); // A walk with names from something that is not a directory is not a walk. got = try h.walkTo(10, 2, &.{"index"}); try testing.expectEqual(@as(u16, 1), got.msg.rwalk.nwqid); try h.send(11, .{ .twalk = .{ .fid = 2, .newfid = 3, .nwname = 1, .wname = wnames(&.{"body"}) } }); got = try h.reap(); try testing.expectEqualStrings(e_not_dir, got.msg.rerror.ename); // A name no fid could hold is refused on its length rather than looked up, // which is what keeps `Rstat`'s name field honest. got = try h.walkTo(12, 4, &.{"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"}); try testing.expectEqualStrings(e_illegal_name, got.msg.rerror.ename); // An in-use `newfid` is refused before anything is walked. try h.send(13, .{ .twalk = .{ .fid = 0, .newfid = 1, .nwname = 1, .wname = wnames(&.{"1"}) } }); got = try h.reap(); try testing.expectEqualStrings(e_fid_in_use, got.msg.rerror.ename); } test "9p server: open then read then clunk, and the release a clunk owes the core" { var h: Harness = .{}; try h.handshake(4096); _ = try h.walkTo(5, 1, &.{ "1", "body" }); try h.send(6, .{ .topen = .{ .fid = 1, .mode = oread } }); var got = try h.reap(); try testing.expectEqual(@as(u64, 18), got.msg.ropen.qid.path); // `iounit` is one message less the slack `fcall.h:72` reserves. try testing.expectEqual(@as(u32, 4096 - iohdrsz), got.msg.ropen.iounit); // The fid IS the open, so a second one has nothing to mean. try h.send(7, .{ .topen = .{ .fid = 1, .mode = oread } }); got = try h.reap(); try testing.expectEqualStrings(e_already_open, got.msg.rerror.ename); try h.send(8, .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); got = try h.reap(); try testing.expectEqualStrings("hello, body\n", got.msg.rread.data); // Offsets are honoured on `body`, which is the whole reason `cat`, `wc` // and `tail` work against this tree (docs/9p.typ §4). try h.send(9, .{ .tread = .{ .fid = 1, .offset = 7, .count = 4096 } }); got = try h.reap(); try testing.expectEqualStrings("body\n", got.msg.rread.data); // Past the end is zero bytes, which is end of file and not an error. try h.send(10, .{ .tread = .{ .fid = 1, .offset = 99, .count = 16 } }); got = try h.reap(); try testing.expectEqual(@as(usize, 0), got.msg.rread.data.len); // THE RELEASE. Without it a pane's `event` reader count never comes back // down and the editor answers to a script that has gone. try testing.expectEqual(@as(u32, 0), h.fsys.releases); try h.send(11, .{ .tclunk = .{ .fid = 1 } }); got = try h.reap(); try testing.expect(got.msg == .rclunk); try testing.expectEqual(@as(u32, 1), h.fsys.releases); // A FID USED AFTER CLUNK is a fid nobody knows. try h.send(12, .{ .tread = .{ .fid = 1, .offset = 0, .count = 16 } }); got = try h.reap(); try testing.expectEqualStrings(e_unknown_fid, got.msg.rerror.ename); try h.send(13, .{ .tclunk = .{ .fid = 1 } }); got = try h.reap(); try testing.expectEqualStrings(e_unknown_fid, got.msg.rerror.ename); // A clunk of a fid that was never opened needs no release at all. _ = try h.walkTo(14, 2, &.{"index"}); try h.send(15, .{ .tclunk = .{ .fid = 2 } }); got = try h.reap(); try testing.expect(got.msg == .rclunk); try testing.expectEqual(@as(u32, 1), h.fsys.releases); } test "9p server: every Rread is clamped to the client's count and to the msize" { var h: Harness = .{}; // A small msize on purpose: `index` is a kilobyte and one message cannot // carry it. try h.handshake(512); _ = try h.walkTo(5, 1, &.{"index"}); try h.send(6, .{ .topen = .{ .fid = 1, .mode = oread } }); _ = try h.reap(); // The count, when the count is the smaller. try h.send(7, .{ .tread = .{ .fid = 1, .offset = 0, .count = 5 } }); var got = try h.reap(); try testing.expectEqual(@as(usize, 5), got.msg.rread.data.len); // The MSIZE, when the client asks for more than one message can hold. An // `Rread` longer than the count is a hard -EIO in Linux // (`client.c:1475-1479`) and one longer than the msize is unreadable, so // the answer is `msize - 11` exactly and the whole message is `msize`. try h.send(8, .{ .tread = .{ .fid = 1, .offset = 0, .count = 1 << 20 } }); const out = h.srv.output(); try testing.expectEqual(@as(?u32, 512), frameLen(out)); got = try h.reap(); try testing.expectEqual(@as(usize, 512 - header_len - 4), got.msg.rread.data.len); // And the core was never asked for bytes that would have been thrown // away: the clamp is on the request too. try h.send(9, .{ .tread = .{ .fid = 1, .offset = 0, .count = 1 << 20 } }); got = try h.reap(); try testing.expectEqual(@as(usize, 501), got.msg.rread.data.len); } test "9p server: a directory read is whole stat records at a cursor the client cannot invent" { var h: Harness = .{}; try h.handshake(4096); // Walked before the root is opened, because an open fid cannot be walked // (walk(5)) and the mode assertion at the bottom of this test needs it. _ = try h.walkTo(4, 1, &.{"index"}); try h.send(5, .{ .topen = .{ .fid = 0, .mode = oread } }); _ = try h.reap(); var found: [8][]const u8 = undefined; // Two entries fit in 150 bytes; the third does not, so it is not counted // and the next read asks for it by index. No cached entry anywhere, which // is what `acmefs`'s re-staging buys (`acmefs.zig:1013-1016`). try h.send(6, .{ .tread = .{ .fid = 0, .offset = 0, .count = 150 } }); var got = try h.reap(); const first = got.msg.rread.data.len; try testing.expectEqual(@as(usize, 2), try dirNames(got.msg.rread.data, &found)); try testing.expectEqualStrings("index", found[0]); try testing.expectEqualStrings("cons", found[1]); try testing.expect(first <= 150); // THE RULE: the next read carries exactly the byte offset where the last // one ended. Not the entry count, and not anything the client chose. try h.send(7, .{ .tread = .{ .fid = 0, .offset = first, .count = 150 } }); got = try h.reap(); const second = got.msg.rread.data.len; try testing.expectEqual(@as(usize, 2), try dirNames(got.msg.rread.data, &found)); try testing.expectEqualStrings("new", found[0]); try testing.expectEqualStrings("1", found[1]); // An arbitrary offset is refused — even one that would land on an entry // boundary, which is exactly the case `ad` accepts silently (`9P-5`). try h.send(8, .{ .tread = .{ .fid = 0, .offset = first + second + 1, .count = 150 } }); got = try h.reap(); try testing.expectEqualStrings(e_bad_offset, got.msg.rerror.ename); try h.send(9, .{ .tread = .{ .fid = 0, .offset = 3, .count = 150 } }); got = try h.reap(); try testing.expectEqualStrings(e_bad_offset, got.msg.rerror.ename); // The refusal did not move the cursor: the listing carries on. try h.send(10, .{ .tread = .{ .fid = 0, .offset = first + second, .count = 150 } }); got = try h.reap(); try testing.expectEqual(@as(usize, 1), try dirNames(got.msg.rread.data, &found)); try testing.expectEqualStrings("2", found[0]); // Zero bytes is END OF DIRECTORY, and it is not an error. try h.send(11, .{ .tread = .{ .fid = 0, .offset = first + second + got.msg.rread.data.len, .count = 150 } }); got = try h.reap(); try testing.expectEqual(@as(usize, 0), got.msg.rread.data.len); // Offset zero REWINDS, which is the only seek 9P allows in a directory. try h.send(12, .{ .tread = .{ .fid = 0, .offset = 0, .count = 150 } }); got = try h.reap(); try testing.expectEqual(@as(usize, 2), try dirNames(got.msg.rread.data, &found)); try testing.expectEqualStrings("index", found[0]); // A count too small for ONE entry is refused rather than answered with // zero bytes, because zero bytes means end of directory and a client that // believes it stops asking. try h.send(13, .{ .tread = .{ .fid = 0, .offset = 0, .count = 40 } }); got = try h.reap(); try testing.expectEqualStrings(e_count_small, got.msg.rerror.ename); // A directory's entries carry the tree's default mode and a zero length; // the exact bits come from `Tstat`, which asks the core. try h.send(14, .{ .tread = .{ .fid = 0, .offset = 0, .count = 150 } }); got = try h.reap(); const one = try Stat.decode(got.msg.rread.data[0 .. std.mem.readInt(u16, got.msg.rread.data[0..2], .little) + 2]); try testing.expectEqualStrings("index", one.name); try testing.expectEqual(@as(u32, dirent_file_perm), one.mode); try testing.expectEqual(@as(u64, 0), one.length); try testing.expectEqual(@as(u32, 0), one.qid.version); try h.send(16, .{ .tstat = .{ .fid = 1 } }); got = try h.reap(); try testing.expectEqual(@as(u32, 0o400), got.msg.rstat.stat.mode); try testing.expectEqual(@as(u64, 1024), got.msg.rstat.stat.length); } test "9p server: a blocked read parks, and the connection keeps working" { var h: Harness = .{}; try h.handshake(4096); _ = try h.walkTo(5, 1, &.{ "1", "event" }); try h.send(6, .{ .topen = .{ .fid = 1, .mode = oread } }); _ = try h.reap(); // `Status.again`: nothing consumed, ask me later. NOT an error and NOT an // empty read — the client hears nothing at all. try h.send(7, .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); try h.quiet(); // A retry round finds it, the core still has nothing, and it goes back. h.pump(); h.pump(); try h.quiet(); // Meanwhile the connection is not blocked: another tag is served while the // read waits, which is the entire point of parking rather than waiting. _ = try h.walkTo(8, 2, &.{ "1", "body" }); try h.send(9, .{ .tstat = .{ .fid = 2 } }); var got = try h.reap(); try testing.expectEqualStrings("body", got.msg.rstat.stat.name); // The core has something now, and the retry round is what delivers it — // with the tag the client used seven messages ago. h.fsys.event = "Kli7 7 0 0 hello\n"; h.pump(); got = try h.reap(); try testing.expectEqual(@as(u16, 7), got.tag); try testing.expectEqualStrings("Kli7 7 0 0 hello\n", got.msg.rread.data); try h.quiet(); // A write the core parks is copied out of the input buffer, so the next // message may overwrite it and the retry still has its bytes. _ = try h.walkTo(10, 3, &.{ "1", "ctl" }); try h.send(11, .{ .topen = .{ .fid = 3, .mode = owrite } }); _ = try h.reap(); h.fsys.park_writes = true; try h.send(12, .{ .twrite = .{ .fid = 3, .offset = 0, .data = "clean\n" } }); try h.quiet(); try h.send(13, .{ .tstat = .{ .fid = 2 } }); _ = try h.reap(); h.fsys.park_writes = false; h.pump(); got = try h.reap(); try testing.expectEqual(@as(u16, 12), got.tag); try testing.expectEqual(@as(u32, 6), got.msg.rwrite.count); try testing.expectEqualStrings("clean\n", h.fsys.writes[0..h.fsys.writes_len]); // Overflow is a refusal and not a queue. Thirty-two blocked readers is // thirty-two scripts watching one session; the thirty-third is told to // retry, which is honest, rather than dropped, which would hang it. for (0..max_slots) |k| { try h.send(@intCast(100 + k), .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); try h.quiet(); } try h.send(200, .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); got = try h.reap(); try testing.expectEqualStrings(e_again, got.msg.rerror.ename); } test "9p server: the reply queue is a FIFO that survives a partial write" { var h: Harness = .{}; try h.handshake(4096); // A UART writes what it can. What is left stays, in order, and the next // reply lands behind it rather than on top of it. try h.send(5, .{ .tstat = .{ .fid = 0 } }); var saved: [256]u8 = undefined; const one = h.srv.output(); const n = one.len; @memcpy(saved[0..n], one); h.srv.wrote(3); try testing.expectEqual(n - 3, h.srv.output().len); try h.send(6, .{ .tstat = .{ .fid = 0 } }); const rest = h.srv.output(); try testing.expectEqualSlices(u8, saved[3..n], rest[0 .. n - 3]); const tail = rest[n - 3 ..]; const second = try decode(tail[0..frameLen(tail).?]); try testing.expectEqual(@as(u16, 6), second.tag); // `push` takes what there is room for and says how much, which is the only // back-pressure a server with no descriptor has. (A buffer of zeros is // also a `size` no encoder produced, so the connection dies on it — which // is the other half of what a caller has to handle.) var flood: [8192]u8 = @splat(0); try testing.expectEqual(@as(usize, 4096), h.srv.push(&flood)); h.pump(); try testing.expect(h.srv.dead); try testing.expectEqual(@as(usize, 0), h.srv.push(&flood)); } test "9p server: Tflush answers the original first and the Rflush second" { var h: Harness = .{}; try h.handshake(4096); _ = try h.walkTo(5, 1, &.{ "1", "event" }); try h.send(6, .{ .topen = .{ .fid = 1, .mode = oread } }); _ = try h.reap(); try h.send(7, .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); try h.quiet(); // TWO messages, in this order and no other. A client that sees `Rflush` // may reuse the tag, so a reply arriving after it would be a reply to // whatever that tag names next. try h.send(8, .{ .tflush = .{ .oldtag = 7 } }); var got = try h.reap(); try testing.expectEqual(@as(u16, 7), got.tag); try testing.expectEqualStrings(e_interrupted, got.msg.rerror.ename); got = try h.reap(); try testing.expectEqual(@as(u16, 8), got.tag); try testing.expect(got.msg == .rflush); try h.quiet(); // The park slot is gone with it: the core producing an event now sends // nothing, rather than a second answer to a tag the client has reused. h.fsys.event = "Kli7 7 0 0 hello\n"; h.pump(); try h.quiet(); // A flush of a tag we do not hold is an `Rflush` and nothing else, which // is the only promise flush(5) makes. try h.send(9, .{ .tflush = .{ .oldtag = 99 } }); got = try h.reap(); try testing.expectEqual(@as(u16, 9), got.tag); try testing.expect(got.msg == .rflush); try h.quiet(); } test "9p server: the fid and permission refusals, each in a string Linux knows" { var h: Harness = .{}; try h.handshake(4096); // A fid nobody walked to. for ([_]Msg{ .{ .tread = .{ .fid = 99, .offset = 0, .count = 16 } }, .{ .tstat = .{ .fid = 99 } }, .{ .tclunk = .{ .fid = 99 } }, .{ .topen = .{ .fid = 99, .mode = oread } }, .{ .twalk = .{ .fid = 99, .newfid = 98, .nwname = 0 } }, }, 20..) |msg, tag| { try h.send(@intCast(tag), msg); const got = try h.reap(); try testing.expectEqualStrings(e_unknown_fid, got.msg.rerror.ename); } // A fid that was never opened, and one opened the other way round. Both // are `u9fs.c:755-758`'s two conditions. _ = try h.walkTo(5, 1, &.{ "1", "body" }); try h.send(6, .{ .tread = .{ .fid = 1, .offset = 0, .count = 16 } }); var got = try h.reap(); try testing.expectEqualStrings(e_bad_use, got.msg.rerror.ename); try h.send(7, .{ .topen = .{ .fid = 1, .mode = oread } }); _ = try h.reap(); try h.send(8, .{ .twrite = .{ .fid = 1, .offset = 0, .data = "x" } }); got = try h.reap(); try testing.expectEqualStrings(e_bad_use, got.msg.rerror.ename); // «must not have been opened for I/O» — walk(5). The fid IS the open. try h.send(9, .{ .twalk = .{ .fid = 1, .newfid = 2, .nwname = 0 } }); got = try h.reap(); try testing.expectEqualStrings(e_bad_use, got.msg.rerror.ename); // THE PERMISSION CHECK IS OURS: there is no kernel above us to do it, and // `errors` is write-only in acme's own dirtab. _ = try h.walkTo(10, 3, &.{ "1", "errors" }); try h.send(11, .{ .topen = .{ .fid = 3, .mode = oread } }); got = try h.reap(); try testing.expectEqualStrings(e_perm, got.msg.rerror.ename); try h.send(12, .{ .topen = .{ .fid = 3, .mode = owrite } }); got = try h.reap(); try testing.expect(got.msg == .ropen); // A directory is read and only read; and nothing here can be removed, so // nothing can be opened remove-on-close or executed either. try h.send(13, .{ .topen = .{ .fid = 0, .mode = ordwr } }); got = try h.reap(); try testing.expectEqualStrings(e_perm, got.msg.rerror.ename); _ = try h.walkTo(14, 4, &.{ "1", "tag" }); for ([_]u8{ orclose, oexec, oread | orclose }, 30..) |mode, tag| { try h.send(@intCast(tag), .{ .topen = .{ .fid = 4, .mode = mode } }); got = try h.reap(); try testing.expectEqualStrings(e_perm, got.msg.rerror.ename); } } test "9p server: Twstat with a zero length is the truncate, and so is OTRUNC" { var h: Harness = .{}; try h.handshake(4096); _ = try h.walkTo(5, 1, &.{ "1", "body" }); // The sentinels stat(5) specifies: an empty string, an all-ones integer. const sentinel: Stat = .{ .type = std.math.maxInt(u16), .dev = std.math.maxInt(u32), .qid = .{ .type = 0xFF, .version = std.math.maxInt(u32), .path = std.math.maxInt(u64) }, .mode = std.math.maxInt(u32), .atime = std.math.maxInt(u32), .mtime = std.math.maxInt(u32), .length = std.math.maxInt(u64), .name = "", .uid = "", .gid = "", .muid = "", }; // Nothing to do: accepted and ignored, which is what a filesystem of live // editor state has to do with a mode, an owner and two times. try h.send(6, .{ .twstat = .{ .fid = 1, .stat = sentinel } }); var got = try h.reap(); try testing.expect(got.msg == .rwstat); try testing.expectEqualStrings("hello, body\n", h.fsys.body); // A length that is neither the sentinel nor zero. The core honours exactly // one value, and this string is one Linux maps to EPERM rather than 526. var five = sentinel; five.length = 5; try h.send(7, .{ .twstat = .{ .fid = 1, .stat = five } }); got = try h.reap(); try testing.expectEqualStrings(e_trunc_only, got.msg.rerror.ename); // A rename would change the shape of a tree that follows the pane list. var renamed = sentinel; renamed.name = "other"; try h.send(8, .{ .twstat = .{ .fid = 1, .stat = renamed } }); got = try h.reap(); try testing.expectEqualStrings(e_wstat, got.msg.rerror.ename); try testing.expectEqualStrings("hello, body\n", h.fsys.body); // ...and zero IS the truncate. var zero = sentinel; zero.length = 0; try h.send(9, .{ .twstat = .{ .fid = 1, .stat = zero } }); got = try h.reap(); try testing.expect(got.msg == .rwstat); try testing.expectEqualStrings("", h.fsys.body); // The other spelling of the same thing, and the one a shell's `>` // produces: `Topen` with `OTRUNC` is a truncate and then an open, in that // order, and it is refused on a fid with no write permission. h.fsys.body = "hello, body\n"; _ = try h.walkTo(10, 2, &.{"index"}); try h.send(11, .{ .topen = .{ .fid = 2, .mode = oread | otrunc } }); got = try h.reap(); try testing.expectEqualStrings(e_perm, got.msg.rerror.ename); try h.send(12, .{ .topen = .{ .fid = 1, .mode = owrite | otrunc } }); got = try h.reap(); try testing.expectEqual(@as(u64, 18), got.msg.ropen.qid.path); try testing.expectEqualStrings("", h.fsys.body); } test "9p server: create and remove are refused, and a remove clunks the fid anyway" { var h: Harness = .{}; try h.handshake(4096); // Nothing in a generated tree is a client's to make. The one place a // client DOES create something is `new/`, where the WALK creates a pane — // so the capability exists and is not spelled `Tcreate` (`9P-18`). try h.send(5, .{ .tcreate = .{ .fid = 0, .name = "thing", .perm = 0o600, .mode = owrite } }); var got = try h.reap(); try testing.expectEqualStrings(e_perm, got.msg.rerror.ename); // A remove is refused too — but «the fid is clunked even if the remove // fails», which is the half of remove(5) that is easy to miss, and the // release still goes to the core. _ = try h.walkTo(6, 1, &.{ "1", "body" }); try h.send(7, .{ .topen = .{ .fid = 1, .mode = ordwr } }); _ = try h.reap(); try h.send(8, .{ .tremove = .{ .fid = 1 } }); got = try h.reap(); try testing.expectEqualStrings(e_perm, got.msg.rerror.ename); try testing.expectEqual(@as(u32, 1), h.fsys.releases); try h.send(9, .{ .tstat = .{ .fid = 1 } }); got = try h.reap(); try testing.expectEqualStrings(e_unknown_fid, got.msg.rerror.ename); } test "9p server: a message arriving a byte at a time is served when its last byte lands" { var h: Harness = .{}; try h.handshake(4096); // The board's UART, and a socket that happened to split a write. Framing // is `size[4]` and nothing may be served until all of it is in. var buf: [64]u8 = undefined; const bytes = try encode(.{ .tstat = .{ .fid = 0 } }, 5, &buf); for (bytes[0 .. bytes.len - 1]) |b| { try testing.expectEqual(@as(usize, 1), h.srv.push(&.{b})); h.pump(); try h.quiet(); } try testing.expectEqual(@as(usize, 1), h.srv.push(bytes[bytes.len - 1 ..])); h.pump(); const got = try h.reap(); try testing.expectEqualStrings("/", got.msg.rstat.stat.name); // TWO messages in one push are two replies, in order, and the input buffer // ends up empty. var pair: [128]u8 = undefined; const a = try encode(.{ .tstat = .{ .fid = 0 } }, 6, &pair); const b = try encode(.{ .tstat = .{ .fid = 0 } }, 7, pair[a.len..]); try testing.expectEqual(a.len + b.len, h.srv.push(pair[0 .. a.len + b.len])); h.pump(); try testing.expectEqual(@as(u16, 6), (try h.reap()).tag); try testing.expectEqual(@as(u16, 7), (try h.reap()).tag); try h.quiet(); } test "9p server: what is not 9P2000 on this connection is refused, not guessed" { var h: Harness = .{}; try h.handshake(4096); var buf: [64]u8 = undefined; const good = try encode(.{ .tstat = .{ .fid = 0 } }, 5, &buf); // An R-message: a client on the wrong end of the connection, or the // double-role link docs/9p.typ §7 says not to build. var raw: [64]u8 = undefined; @memcpy(raw[0..good.len], good); raw[4] = @intFromEnum(Type.rstat); try testing.expectEqual(good.len, h.srv.push(raw[0..good.len])); h.pump(); var got = try h.reap(); try testing.expectEqual(@as(u16, 5), got.tag); try testing.expectEqualStrings(e_botch, got.msg.rerror.ename); // A type byte no dialect we serve defines — 8 is 9P2000.L's `Tstatfs` — // still gets an answer, because the tag is at a fixed offset and a client // that gets no reply hangs. @memcpy(raw[0..good.len], good); raw[4] = 8; _ = h.srv.push(raw[0..good.len]); h.pump(); got = try h.reap(); try testing.expectEqual(@as(u16, 5), got.tag); try testing.expectEqualStrings(e_botch, got.msg.rerror.ename); // A `size` no encoder could have produced is not a message to answer: the // stream is not 9P and there is no resynchronising from it. @memcpy(raw[0..good.len], good); std.mem.writeInt(u32, raw[0..4], 3, .little); _ = h.srv.push(raw[0..good.len]); h.pump(); try h.quiet(); try testing.expect(h.srv.dead); } test "9p server: a connection that drops still pays the core its releases" { var h: Harness = .{}; try h.handshake(4096); _ = try h.walkTo(5, 1, &.{ "1", "event" }); _ = try h.walkTo(6, 2, &.{ "1", "body" }); for ([_]u32{ 1, 2 }, 7..) |fid, tag| { try h.send(@intCast(tag), .{ .topen = .{ .fid = fid, .mode = oread } }); _ = try h.reap(); } // One blocked reader, so there is a parked request to abandon as well. try h.send(9, .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); try h.quiet(); // The socket died. Every open fid still owes the core a release, and that // debt outlives the connection — losing it leaves the editor reporting // button actions to a script that is gone. h.srv.hangup(); h.pump(); try testing.expectEqual(@as(u32, 2), h.fsys.releases); // ...and nothing is written to a socket that has gone. try h.quiet(); // `Tversion` is the same reset on a live connection: fids clunked, // outstanding I/O abandoned, releases still paid (version(5)). var g: Harness = .{}; try g.handshake(4096); _ = try g.walkTo(5, 1, &.{ "1", "event" }); try g.send(6, .{ .topen = .{ .fid = 1, .mode = oread } }); _ = try g.reap(); try g.send(7, .{ .tread = .{ .fid = 1, .offset = 0, .count = 4096 } }); try g.quiet(); try g.send(notag, .{ .tversion = .{ .msize = 4096, .version = "9P2000" } }); const v = try g.reap(); try testing.expectEqualStrings("9P2000", v.msg.rversion.version); try testing.expectEqual(@as(u32, 1), g.fsys.releases); // The abandoned read is never answered, and the fid is gone. g.fsys.event = "Kli7 7 0 0 hello\n"; g.pump(); try g.quiet(); try g.send(8, .{ .tstat = .{ .fid = 1 } }); const got = try g.reap(); try testing.expectEqualStrings(e_unknown_fid, got.msg.rerror.ename); } test "9p server: every errno the core can answer is a string Linux knows" { // The nine values `acmefs.E` defines, spelled exactly as // `linux/net/9p/error.c:41-171` holds them. A typo in any of these is // "Unknown error 526" on every `mount -t 9p`, which is why they are // asserted as literals rather than derived from anything. try testing.expectEqualStrings("Operation not permitted", errString(1)); try testing.expectEqualStrings("No such file or directory", errString(2)); try testing.expectEqualStrings("Input/output error", errString(5)); try testing.expectEqualStrings("Cannot allocate memory", errString(12)); try testing.expectEqualStrings("Not a directory", errString(20)); try testing.expectEqualStrings("Invalid argument", errString(22)); try testing.expectEqualStrings("Too many open files in system", errString(23)); try testing.expectEqualStrings("No space left on device", errString(28)); try testing.expectEqualStrings("Function not implemented", errString(38)); // A number this file never emits is EIO, not a table miss. try testing.expectEqualStrings("Input/output error", errString(0)); try testing.expectEqualStrings("Input/output error", errString(999)); // The server's own strings, from the same table and the fossil/u9fs half // of it. Every one of these is a line in `error.c`, which is the whole // difference between an errno and 526. try testing.expectEqualStrings("fid unknown or out of range", e_unknown_fid); try testing.expectEqualStrings("fid already in use", e_fid_in_use); try testing.expectEqualStrings("bad use of fid", e_bad_use); try testing.expectEqualStrings("bad offset in directory read", e_bad_offset); try testing.expectEqualStrings("permission denied", e_perm); try testing.expectEqualStrings("not a directory", e_not_dir); try testing.expectEqualStrings("file already open for I/O", e_already_open); try testing.expectEqualStrings("illegal name", e_illegal_name); try testing.expectEqualStrings("Too many open files in system", e_too_many_fids); try testing.expectEqualStrings("protocol botch", e_botch); try testing.expectEqualStrings("Interrupted system call", e_interrupted); try testing.expectEqualStrings("only support truncation to zero length", e_trunc_only); try testing.expectEqualStrings("wstat prohibited", e_wstat); try testing.expectEqualStrings("Resource temporarily unavailable", e_again); // Every one of them fits the buffer a Plan 9 client has for it. for ([_][]const u8{ e_unknown_fid, e_fid_in_use, e_bad_use, e_bad_offset, e_perm, e_not_dir, e_already_open, e_botch, e_interrupted, e_trunc_only, e_wstat, e_again, e_no_tree, e_no_auth, e_count_small, e_illegal_name, e_too_many_fids, e_small_msize, }) |s| try testing.expect(s.len <= errmax); } test "9p server: the fid table and the park table are what the board was costed for" { // `docs/registry.typ` `9P-11` costed a fid table at 32 x 16 bytes. The // real entry is larger, and the difference is not a mistake in either // place: it is the entry NAME, which `Rstat` carries and a node id does // not, plus the open handle and the two-coordinate directory cursor. The // numbers are asserted here so that a change to `Fid` shows up as a diff // in the board's budget rather than as a surprise on the board. // // TWO budgets, because there are now two numbers. `max_fids` is the desktop // one and it is sized for a MOUNT, which keeps a fid per cached inode; // `board_fids` is what a microcontroller serving its own small tree uses, // and it is the one `9P-11` costed. const S = Server(StubFs); const entry = @sizeOf(S.Fid); const slots = @sizeOf(S.Slot) * max_slots; try testing.expect(slots <= 8 * 1024); // The board: two buffers at a 4,096-byte msize — `in` and `out`'s two — // plus the two tables, against 336 KB of free heap on the P4. const board = entry * board_fids; try testing.expect(board <= 3 * 1024); try testing.expect(3 * 4096 + board + slots <= 24 * 1024); // The desktop, against the ≈34 KiB per connection `fs9_service` budgets. // Eight times the fids is ≈16 KiB, and it is the price of `find` working // over a `9pfuse` mount — see `max_fids`. try testing.expect(entry * max_fids <= 24 * 1024); } // --------------------------------------------------------------------------- // the client // --------------------------------------------------------------------------- /// Tags one client may have outstanding at once. /// /// SIXTEEN, and the reasoning is `max_fids`': a fixed array with no allocator, /// scanned rather than mapped, and a refusal rather than a queue when it fills. /// The protocol's tag space is 0..0xFFFE — `notag` is 0xFFFF and belongs to the /// handshake — so this uses the bottom sixteen of sixty-five thousand and never /// a number above them. THE TAG IS ITS OWN INDEX, which is what makes matching /// a reply O(1) with no search and no bookkeeping: see `tags`. /// /// MANY OUTSTANDING REQUESTS ARE LEGAL and the prior art imposes no bound at /// all: Linux's client takes a tag per request out of an IDR /// (`net/9p/client.c:194-199`) and Plan 9's devmnt keeps an `Mntrpc` per /// request on a free list (`devmnt.c:783-800`), because on both the outstanding /// count is «one per process blocked in an I/O», which the kernel already /// bounds elsewhere. Here the count is "one per thing pardes is fetching", and /// a screen does not hold sixteen remote panes. Overflow is `error.NoTags` at /// the moment of asking — the caller collects an answer and asks again — and /// never a silent wait, because a client that blocks is the one thing this /// design does not have anywhere to put. pub const max_tags: usize = 16; /// `Twrite`'s own header: `size[4] type[1] tag[2] fid[4] offset[8] count[4]`. /// What `maxWrite` subtracts from the msize. /// /// NOT `iohdrsz`. That number (24) is the slack a SERVER quotes in `iounit` and /// is deliberately larger than any real header; a client sizing its own request /// against it leaves a byte on the table on every write forever. const twrite_header: usize = header_len + 4 + 8 + 4; /// `Rread`'s header: `size[4] type[1] tag[2] count[4]`. What `maxRead` /// subtracts, and the reason a client's `count` is not simply the msize — the /// reply has to carry a header too, and a `count` of msize is a reply eleven /// bytes too long for the connection that asked for it. const rread_header: usize = header_len + 4; comptime { assert(twrite_header == 23); assert(rread_header == 11); // The tag IS the index, so the table's length is the tag space in use, and // the handshake's `notag` must fall outside it or a `Rversion` would land // on somebody's slot. assert(max_tags <= notag); // At the smallest msize this file will agree to, a read and a write must // both still be able to carry a byte, or a connection could be negotiated // that cannot do any I/O at all. assert(msize_min > rread_header); assert(msize_min > twrite_header); } /// Every way `submit` can refuse, and each is a different thing for the caller /// to do about it. pub const ClientError = error{ /// All `max_tags` are outstanding. Collect an answer and ask again. NoTags, /// The out queue has no room for this request. Write `output()` out and /// ask again. THE ONLY BACK-PRESSURE a sans-io client has. NoSpace, /// The request cannot fit the negotiated msize: a `Twrite` past /// `maxWrite()`, a `Tread` asking past `maxRead()`, a sixteen-element walk /// of long names. REFUSED AND NOT CLAMPED, because `submit` answers with a /// tag and nothing else — a silent clamp would leave the caller to guess /// how much of its buffer went, and guess wrong about the offset to /// continue from. `maxRead` and `maxWrite` are how a caller chunks first. TooLarge, /// A request before `Rversion` has landed, or a second `Tversion` while /// requests are outstanding. Handshake, /// The stream is not 9P any more and this connection is finished. See /// `dead`. Dead, /// A request no encoding of 9P admits: `nofid` as a fid, an empty walk /// element or one with a separator in it, more than `max_welem` elements. /// A bug in the caller, caught here rather than spent as a round trip. BadRequest, }; /// A 9P2000 client for one connection. /// /// THE MIRROR OF `Server`, and deliberately the same shape: caller-owned `in` /// and `out` buffers, no allocator, no threads, no descriptor, `std` for /// `readInt`/`writeInt` and nothing else. So it compiles for the board's /// `riscv32-freestanding` and runs over `src/esp32p4/uart.zig`'s non-blocking /// receive and bounded-spin transmit exactly as it runs over a unix socket in /// `src/fs9_client.zig` — which is the whole reason for the shape, because a /// client that owned its descriptor would be a client that could not. /// /// THE API IS A STATE MACHINE AND NOT `fn read() []u8`, because the core is /// single-threaded and never blocks (docs/9p.typ §12.5). The three moving parts /// are: /// /// 1. `submit(Request)` ENCODES a T-message into the out queue and hands /// back its tag. It never waits and never touches a descriptor; a full /// queue or a full tag table is a refusal the caller can act on. /// 2. `push`/`output`/`wrote` move bytes, in whatever sizes the transport /// manages, in whatever order they arrive. /// 3. `take()` answers with the next COMPLETED operation, or null when there /// is not a whole reply buffered yet. A caller's frame is /// `while (client.take()) |done| ...`, which is `Server.next()`'s own /// loop-until-null contract read from the other side. /// /// WHY COMPLETION IS A PULL AND NOT A CALLBACK: a callback would run inside /// `push`, which is inside the transport's read, which is inside the host's /// poll dispatch — and `src/fs9_service.zig` already states why filesystem /// work must not happen there. Pulling puts the caller's own code back on the /// caller's own stack. /// /// WHY THERE IS NO PER-TAG RESULT QUEUE: one frame completes exactly one /// operation, and `take` returns it immediately, so there is never a completed /// answer nobody has collected. That is what keeps a tag slot two bytes wide /// instead of an msize wide, and it is why `Done` may borrow `in` (see there). /// /// REPLIES MAY ARRIVE IN ANY ORDER and this client does not care: the tag is /// its own index into `tags`, so attribution is one bounds check and one /// array read, with no assumption about arrival order anywhere in the file. /// The reply's TYPE is checked against the request's `Op` as well, because a /// tag is only as good as the table behind it. /// /// MEMORY: the two buffers, and `@sizeOf(Client)` for everything else — a /// sixteen-entry tag table of eight-byte entries plus nine scalars, asserted at /// the bottom of this file. Nothing here grows and nothing here is allocated. pub const Client = struct { /// Reply bytes the caller has pushed. `in[0..frame]` is the reply most /// recently returned by `take`, and every slice a `Done` holds points into /// it — which is why nothing compacts this buffer until the next `take`. in: []u8, /// Encoded requests, oldest first, as a byte FIFO. Every 9P message /// carries its own length, so the queue needs no side table. out: []u8, in_len: usize = 0, frame: u32 = 0, out_len: usize = 0, out_off: usize = 0, /// Negotiated by the handshake; ZERO means "not on a protocol yet", and /// nothing but `version` may be submitted in that state. It is also zero /// after an `Rversion` of "unknown", which is a completed handshake with /// no dialect in common. msize: u32 = 0, /// What our own `Tversion` offered, kept only so that `Rversion` can be /// checked against it: «the server responds with its own maximum, which /// must be less than or equal to the client's». asked: u32 = 0, /// A `Tversion` is outstanding. Its tag is `notag`, so it cannot live in /// the table below — and it does not need to, because the protocol allows /// nothing else to be outstanding beside it. versioning: bool = false, /// The stream is not 9P and there is no resynchronising from it. Write-once, /// like `Server.dead`: a reply that cannot be attributed is worse than a /// closed connection, because the caller would wait on it forever. dead: bool = false, /// THE TAG TABLE, indexed BY THE TAG. `tags[t].op` is null when tag `t` is /// free, which makes claiming a tag a scan of sixteen and matching a reply /// a single index — and it means a caller may keep its own per-request /// state in a plain sixteen-entry array of its own, keyed the same way, /// with no map on either side. tags: [max_tags]Slot = @splat(.{}), /// What is remembered about one outstanding request, which is as little as /// the protocol lets us get away with: what it was, and — for a read — /// what it asked for, because `read(5)` bounds the reply by it and a /// server that ignores that bound is handing back bytes at offsets we /// never asked about. const Slot = struct { op: ?Op = null, count: u32 = 0, }; /// What a client asked for. The tag names of `Request` and of `Result`'s /// answers are these, so nothing maps one to the other by hand. /// /// EIGHT OPERATIONS AND NOT THIRTEEN, and the five absences are decisions: /// /// * `Tauth`: there is no authentication in this design and the server half /// refuses it by name (`e_no_auth`). The socket's permissions are the /// protection. /// * `Tcreate`/`Tremove`: the server refuses both, because the shape of the /// tree follows the pane list. Walking into `new/` is how a client creates /// a pane, and that is a `walk`. /// * `Twstat`: the one wstat the tree honours is a truncate, and a client /// that wants to empty a file opens it `OTRUNC` in the same round trip. /// * `Tflush`: nothing here has a cancel button. A flush costs a second tag /// and brings a reply-ORDER rule with it — «the Rflush must come after the /// original reply» — which is a rule nobody exercises if no caller can /// change its mind, and an unexercised ordering rule in a protocol client /// is a bug waiting for its first user. pub const Op = enum { version, attach, walk, open, read, write, clunk, stat }; /// One request, as its caller states it. A `union(Op)` rather than eight /// functions so that `submit` is one entry point with one refusal path: every /// bound this client has — the tag table, the out queue, the msize — applies to /// all eight identically, and a ninth operation cannot forget one of them. /// /// NO TAG FIELD: the tag is what `submit` HANDS BACK. A caller that chose its /// own tags would be maintaining the table this file already maintains. pub const Request = union(Op) { /// The handshake. `msize` is the largest message this client will send or /// accept, and ZERO means "as much as my buffers hold", which is the /// answer a caller with no opinion wants. Clamped to the buffers either /// way; see `beginVersion`. version: struct { msize: u32 = 0 }, /// `afid` is not a parameter: it is always `nofid`, because this client /// never sends `Tauth`. attach: struct { fid: u32, uname: []const u8, aname: []const u8 = "" }, /// The path elements, already split. A SLICE OF SLICES rather than the /// codec's fixed `[max_welem]` array, because a caller has a path and not /// an array: the copy into the fixed array happens once, in `submit`, /// where the `nwname` bound is checked anyway. An empty list is the legal /// zero-element walk, which clones `fid` onto `newfid`. walk: struct { fid: u32, newfid: u32, names: []const []const u8 }, open: struct { fid: u32, mode: u8 }, /// `count` is refused rather than clamped above `maxRead()`; see there. read: struct { fid: u32, offset: u64, count: u32 }, /// `data` is COPIED into the out queue by `submit` and is not borrowed /// afterwards, which is what lets a caller write out of a buffer it is /// about to reuse. write: struct { fid: u32, offset: u64, data: []const u8 }, clunk: struct { fid: u32 }, stat: struct { fid: u32 }, }; /// What one request came to. The answer's SHAPE, which is what a caller acts /// on; `Done.op` says which request it belongs to and `Done.tag` says which /// one of several. pub const Result = union(enum) { /// The server said no: `Rerror`'s string, and the only variant that can /// answer ANY of the eight. Borrows the input buffer — see `Done`. fail: []const u8, /// `version` is "9P2000", or the literal "unknown", which is a SUCCESSFUL /// reply meaning no dialect in common. `Client.msize` is nonzero only in /// the first case, so the second leaves a connection on which nothing can /// be submitted and the caller hangs up. version: struct { msize: u32, version: []const u8 }, attach: Qid, /// `nwqid` may be SHORTER than the walk's element count: a partial walk is /// a success with fewer qids, and only a failure on the FIRST element is /// an `Rerror`. So a caller MUST compare `nwqid` against what it asked for /// before believing its fid landed anywhere. /// /// The whole array is carried rather than only the last qid, because the /// last one is the only thing THIS tree's clients want and the /// intermediate ones are what a caching client caches against /// (`Qid.version`). Two hundred and eight bytes, on a value the caller /// consumes and drops. walk: struct { nwqid: u16, wqid: [max_welem]Qid }, open: struct { qid: Qid, iounit: u32 }, /// The bytes, borrowing the input buffer — see `Done`. SHORTER than the /// requested count is normal and is not the end of the file; ZERO bytes is /// the end of the file. read: []const u8, /// The count actually written, which may be short — the caller advances /// its offset by this and not by what it asked. write: u32, clunk: void, /// Borrows the input buffer for its four strings — see `Done`. stat: Stat, }; /// One completed operation. /// /// BORROWS THE INPUT BUFFER, and this is the whole lifetime rule: a `Done` is /// valid until the next call to anything on the `Client` that produced it. The /// `fail` string, the `read` bytes and the `stat` strings all point into /// `Client.in`, exactly as `decode`'s do and for the same reason — the /// alternative is a per-tag copy of every payload, which on the board is /// sixteen msizes of static RAM to save a caller one `@memcpy` it may not even /// want. `take` releases the previous answer's frame on entry, so the rule is /// enforced by construction rather than by hope: a caller that keeps a `Done` /// across a second `take` is reading bytes the next reply has been decoded /// into. pub const Done = struct { /// The tag `submit` handed out, or `notag` for the handshake. FREE again /// the moment this is returned, so a caller that indexes its own /// sixteen-entry table by tag must read this entry out before submitting /// anything else. tag: u16, /// Which of the eight this answers. Needed beside `result` because /// `Rerror` answers all of them and carries no hint of which. op: Op, result: Result, }; pub const Options = struct { /// Room for one whole reply. Caps the msize with `out`. in: []u8, /// Room for one whole request, at least. MORE room is what buys /// pipelining: sixteen outstanding `Tread`s are sixteen small messages /// that all have to fit here at once, and `submit` answers /// `error.NoSpace` rather than blocking when they do not. out: []u8, }; /// The buffers are the caller's, which is what "no allocator" means from /// this side: the board hands over two static arrays, a host hands over /// two heap slices, and this file cannot tell the difference. The msize /// follows from them and from the server's `Rversion`. pub fn init(opts: Options) Client { assert(opts.in.len >= msize_min); assert(opts.out.len >= msize_min); return .{ .in = opts.in, .out = opts.out }; } /// The connection went away, or the caller is done with it. Unlike /// `Server.hangup` there is no debt to pay: a client owes the far end /// nothing on the way out — its fids are the server's to clean up when the /// stream closes, which is exactly what `Server.hangup` is for. pub fn hangup(c: *Client) void { c.dead = true; c.tags = @splat(.{}); c.versioning = false; c.msize = 0; c.asked = 0; c.in_len = 0; c.frame = 0; c.out_len = 0; c.out_off = 0; } // -- bytes in, bytes out --------------------------------------------- // // The four `Server` has, written out again rather than shared. They look // identical and they are not the same three lines: `Server.push` refuses // once dead, and `Server.hasRoom` reserves a whole msize before a request // is handed to the core so that no reply can fail to be written. A client // reserves nothing — it refuses at `submit`, where the caller is standing // right there — so a shared FIFO would be one struct with two callers and // two exceptions, which is more to read than this is. /// Take as much of `bytes` as there is room for, and answer how much. A /// short answer is not a loss: it is back-pressure, and the caller /// re-offers the tail after `take`ing what it can. Bytes are APPENDED, so /// the reply currently being borrowed by a `Done` does not move. pub fn push(c: *Client, bytes: []const u8) usize { if (c.dead) return 0; const n = @min(bytes.len, c.in.len - c.in_len); @memcpy(c.in[c.in_len..][0..n], bytes[0..n]); c.in_len += n; return n; } /// The requests waiting to go, oldest first, as one contiguous run of /// whole 9P messages. Valid until the next call to anything else here. pub fn output(c: *const Client) []const u8 { return c.out[c.out_off..c.out_len]; } /// How many of `output()`'s bytes actually left. A partial write is normal /// on a UART and on a full socket, and the remainder stays put. pub fn wrote(c: *Client, n: usize) void { assert(n <= c.out_len - c.out_off); c.out_off += n; if (c.out_off == c.out_len) { c.out_off = 0; c.out_len = 0; } } /// Slide the unwritten tail down. Called only when room is wanted, so the /// common case — a fully written queue, reset to empty by `wrote` — never /// moves a byte. fn compact(c: *Client) void { assert(c.out_off <= c.out_len); const n = c.out_len - c.out_off; std.mem.copyForwards(u8, c.out[0..n], c.out[c.out_off..c.out_len]); c.out_off = 0; c.out_len = n; } /// Release the reply `take` last returned and slide the rest of the input /// down. One message-long move per message; a ring buffer would let a /// decoded reply straddle the wrap and stop being one slice. fn dropFrame(c: *Client) void { assert(c.frame != 0); assert(c.frame <= c.in_len); const n = c.frame; std.mem.copyForwards(u8, c.in[0 .. c.in_len - n], c.in[n..c.in_len]); c.in_len -= n; c.frame = 0; } // -- what a caller may ask for --------------------------------------- /// The largest `Tread.count` this connection can answer, which is the /// msize less `Rread`'s own header. Zero before the handshake. pub fn maxRead(c: *const Client) u32 { if (c.msize == 0) return 0; return c.msize - @as(u32, @intCast(rread_header)); } /// The most bytes one `Twrite` can carry, which is the msize less /// `Twrite`'s own header. Zero before the handshake. A caller with more /// than this chunks; see `ClientError.TooLarge` for why it is not clamped. pub fn maxWrite(c: *const Client) u32 { if (c.msize == 0) return 0; return c.msize - @as(u32, @intCast(twrite_header)); } /// Requests outstanding, the handshake included. What a caller's loop /// tests to know whether there is anything left to wait for. pub fn pending(c: *const Client) usize { var n: usize = @intFromBool(c.versioning); for (c.tags) |t| n += @intFromBool(t.op != null); return n; } // -- asking ------------------------------------------------------------ /// Encode one request into the out queue and hand back its tag. Never /// blocks, never waits, never touches a descriptor. /// /// The refusals are in one order on purpose: what is wrong with the /// REQUEST first, then what is wrong with this client's tables, so a /// caller's bad argument never costs a tag and never half-fills the queue. pub fn submit(c: *Client, req: Request) ClientError!u16 { if (c.dead) return error.Dead; if (req == .version) return c.beginVersion(req.version.msize); // «The client must communicate the version before any other messages» // — and until `Rversion` has landed there is no msize to bound // anything by, which is the same gate `Server.startFrame` applies from // the other side. if (c.msize == 0 or c.versioning) return error.Handshake; const msg: Msg = switch (req) { .version => unreachable, // handled above .attach => |m| blk: { if (m.fid == nofid) return error.BadRequest; break :blk .{ .tattach = .{ .fid = m.fid, .afid = nofid, .uname = m.uname, .aname = m.aname, } }; }, .walk => |m| blk: { if (m.fid == nofid or m.newfid == nofid) return error.BadRequest; // `MAXWELEM` is a hard protocol bound and not a buffer size: // every implementation refuses a seventeen-element walk, so a // caller with a deeper path splits it into two walks. if (m.names.len > max_welem) return error.BadRequest; var w: [max_welem][]const u8 = @splat(""); for (m.names, 0..) |n, i| { // An empty element, or one with a separator in it, is a // caller that has not split its path. `Twalk` has no // encoding for either and a server answers the first one // `illegal name` — a round trip spent on a bug that was // visible from here. if (n.len == 0) return error.BadRequest; if (std.mem.indexOfAny(u8, n, "/\x00") != null) return error.BadRequest; w[i] = n; } break :blk .{ .twalk = .{ .fid = m.fid, .newfid = m.newfid, .nwname = @intCast(m.names.len), .wname = w, } }; }, .open => |m| blk: { if (m.fid == nofid) return error.BadRequest; break :blk .{ .topen = .{ .fid = m.fid, .mode = m.mode } }; }, .read => |m| blk: { if (m.fid == nofid) return error.BadRequest; // The one bound the request's own length does not express: // what comes BACK has to fit the connection too. if (m.count > c.maxRead()) return error.TooLarge; break :blk .{ .tread = .{ .fid = m.fid, .offset = m.offset, .count = m.count } }; }, .write => |m| blk: { if (m.fid == nofid) return error.BadRequest; break :blk .{ .twrite = .{ .fid = m.fid, .offset = m.offset, .data = m.data } }; }, .clunk => |m| blk: { if (m.fid == nofid) return error.BadRequest; break :blk .{ .tclunk = .{ .fid = m.fid } }; }, .stat => |m| blk: { if (m.fid == nofid) return error.BadRequest; break :blk .{ .tstat = .{ .fid = m.fid } }; }, }; // ONE ceiling for every request, which is what makes a `Twrite` and a // sixteen-element `Twalk` obey the same rule: the msize is «the // maximum length, in bytes, ... including the size field», and a // client that sends more is a client the server closes on. const need = totalLen(msg) catch return error.TooLarge; if (need > c.msize) return error.TooLarge; const op = std.meta.activeTag(req); const tag = c.claim(op) orelse return error.NoTags; errdefer c.tags[tag] = .{}; try c.emit(tag, msg); if (op == .read) c.tags[tag].count = req.read.count; return tag; } /// `Tversion`, which is the one exchange with no tag and no msize behind /// it. /// /// A SECOND ONE IS A CONNECTION RESET — «all fids are clunked and any /// outstanding I/O is abandoned» (`version(5)`) — and abandoning somebody /// else's request is not this function's decision to make. So it is /// refused while anything is outstanding, and a caller that means to reset /// collects its answers or hangs up first. fn beginVersion(c: *Client, want: u32) ClientError!u16 { if (c.pending() != 0) return error.Handshake; // Two ceilings and the smaller wins: what one reply buffer holds, and // what one request buffer holds. A caller with no opinion passes zero // and gets both. const cap: u32 = @intCast(@min(c.in.len, c.out.len, std.math.maxInt(u32))); const m = @min(if (want == 0) cap else want, cap); // Below the floor there is a connection that cannot carry an `Rwalk`, // which is to say no connection at all. `init` asserts the buffers // clear it, so this can only be a `want` the caller chose. if (m < msize_min) return error.BadRequest; try c.emit(notag, .{ .tversion = .{ .msize = m, .version = "9P2000" } }); c.msize = 0; c.asked = m; c.versioning = true; return notag; } /// The lowest free tag, marked used. Lowest rather than round-robin so /// that a client with one request outstanding always uses tag 0, which /// makes a wire trace readable by eye. fn claim(c: *Client, op: Op) ?u16 { for (&c.tags, 0..) |*t, i| { if (t.op != null) continue; t.* = .{ .op = op }; return @intCast(i); } return null; } /// Queue one request. The only way this fails is room: `totalLen` has /// already refused every other way `encode` can, which is why the error /// set collapses to one value here. fn emit(c: *Client, tag: u16, msg: Msg) ClientError!void { if (c.out_off != 0) c.compact(); const bytes = encode(msg, tag, c.out[c.out_len..]) catch return error.NoSpace; c.out_len += bytes.len; } // -- collecting -------------------------------------------------------- /// The next completed operation, or null when there is not a whole reply /// buffered yet. Call in a loop until null, once per frame. /// /// A `Done` BORROWS the input buffer and is valid until the next call /// here: the previous reply's frame is released on entry, which is what /// makes that rule mechanical instead of a note somebody has to remember. pub fn take(c: *Client) ?Done { if (c.frame != 0) c.dropFrame(); if (c.dead) return null; const len = frameLen(c.in[0..c.in_len]) orelse return null; // A `size` no encoder produced, or one this connection could never // buffer: either way the stream is not 9P and waiting for the rest of // it is waiting forever. if (len < header_len or len > c.in.len) return c.die(); // And a server that sends past the msize it agreed to has stopped // speaking the protocol it agreed to. if (c.msize != 0 and len > c.msize) return c.die(); if (len > c.in_len) return null; c.frame = len; // A body this codec refuses is not a message we can attribute to a // tag, so there is nobody to report it to. The connection ends. const got = decode(c.in[0..len]) catch return c.die(); return c.consume(got); } /// The stream is finished. Returns null so that every refusal in `take` /// and `consume` is one expression. fn die(c: *Client) ?Done { c.dead = true; return null; } /// One decoded reply onto the request it answers. fn consume(c: *Client, got: Decoded) ?Done { // A CLIENT READS R-MESSAGES. A T-message here is the other end of the // connection talking, or the double-role link docs/9p.typ §7 tells us // not to build — the exact mirror of `Server.startFrame`'s refusal, // and the encoding's own parity does the work in both directions. if (isT(got.msg.msgType())) return c.die(); if (got.msg == .rversion) return c.version(got); // Nothing may arrive before a `Tversion` has been answered, and // nothing but the `Rversion` while one is outstanding. if (c.versioning or c.msize == 0) return c.die(); // The tag is its own index, so this bounds check IS the lookup. if (got.tag >= max_tags) return c.die(); const slot = &c.tags[got.tag]; const op = slot.op orelse return c.die(); const result: Result = switch (got.msg) { // `Rerror` answers ANY of the eight, which is exactly why `Done` // reports the op beside it: the string does not say what failed. .rerror => |m| .{ .fail = m.ename }, .rattach => |m| if (op != .attach) return c.die() else .{ .attach = m.qid }, .rwalk => |m| if (op != .walk) return c.die() else .{ .walk = .{ .nwqid = m.nwqid, .wqid = m.wqid }, }, .ropen => |m| if (op != .open) return c.die() else .{ .open = .{ .qid = m.qid, .iounit = m.iounit }, }, .rread => |m| blk: { if (op != .read) return c.die(); // «count ... indicates the number of bytes returned», and // read(5) makes it no more than what was asked. Linux calls a // longer one a hard `-EIO` (`net/9p/client.c:1475-1479`); here // it ends the connection, because the byte after the ones we // asked for is a byte we have no offset to put anywhere. if (m.data.len > slot.count) return c.die(); break :blk .{ .read = m.data }; }, .rwrite => |m| if (op != .write) return c.die() else .{ .write = m.count }, .rclunk => if (op != .clunk) return c.die() else .clunk, .rstat => |m| if (op != .stat) return c.die() else .{ .stat = m.stat }, // The rest are replies to requests this client does not send — // `Rauth`, `Rcreate`, `Rremove`, `Rwstat`, `Rflush` — and one is // not an answer at all (`Rerror`'s illegal twin `Terror`, already // refused by parity above). A reply to a request nobody made means // the tag space is not what we think it is. else => return c.die(), }; slot.* = .{}; return .{ .tag = got.tag, .op = op, .result = result }; } /// `Rversion`: the msize handshake, from the client's side. fn version(c: *Client, got: Decoded) ?Done { if (!c.versioning) return c.die(); // «Rversion ... carries the same tag», and that tag is `notag`, // because tags do not mean anything yet. if (got.tag != notag) return c.die(); const m = got.msg.rversion; // «The server responds with its own maximum, which must be less than // or equal to the client's» — `version(5)`. A larger one is a message // we cannot buffer, and Linux refuses it for that reason // (`net/9p/client.c:840-843`). if (m.msize > c.asked or m.msize < msize_min) return c.die(); c.versioning = false; if (std.mem.eql(u8, m.version, "9P2000")) { c.msize = m.msize; } else if (!std.mem.eql(u8, m.version, "unknown")) { // The reply must be a version the client offered, or "unknown". // Anything else — "9P2000.u", "9P2000.L", a typo — is a server // answering a question we did not ask, and agreeing to a dialect // this file does not implement is how a client sends a `Tattach` // whose layout the other end reads differently. return c.die(); } // "unknown" leaves `msize` at zero: a completed handshake with no // dialect in common, on which nothing can be submitted. The CALLER // decides whether that is worth hanging up over, which is the honest // place for it — a fallback ladder of dialects is a policy and this is // a codec. return .{ .tag = notag, .op = .version, .result = .{ .version = .{ .msize = m.msize, .version = m.version }, } }; } }; // --------------------------------------------------------------------------- // client tests // --------------------------------------------------------------------------- // // Driven against the SERVER IN THIS FILE, in process, over two pairs of // buffers. That is the strongest test available here and it needs no socket: // every byte the client encodes is a byte the server decodes and vice versa, // so a disagreement about a layout, a length or a tag fails a test rather than // waiting for a live daemon. The stub filesystem is the server tests' own, so // the tree the client walks is the tree those tests already pin down. /// One connection with a client at each end of it. Four buffers, because each /// side owns its own two and neither may see the other's. /// /// `srv.out` is twice the msize because `Server` requires it; `cli.out` is not, /// because a client reserves nothing — see `Client.Options`. const Pair = struct { srv_in: [4096]u8 = undefined, srv_out: [8192]u8 = undefined, cli_in: [4096]u8 = undefined, cli_out: [4096]u8 = undefined, fsys: StubFs = .{}, srv: Srv = undefined, cli: Client = undefined, /// The buffers are fields, so neither end can be built until the pair has /// an address. fn start(p: *Pair) void { p.srv = Srv.init(.{ .in = &p.srv_in, .out = &p.srv_out, .root = 1 }); p.cli = Client.init(.{ .in = &p.cli_in, .out = &p.cli_out }); } fn answer(p: *Pair, req: StubFs.Req) void { const a = p.fsys.handle(req); p.srv.reply(&a.reply, a.bytes); } /// THE WIRE: every byte both ways, and each side given every chance to /// work, until nothing moves. A real transport does this a chunk at a time /// in a poll loop; the tests that care about that drip bytes by hand. fn wire(p: *Pair) void { var moved = true; while (moved) { moved = false; while (p.cli.output().len != 0) { const n = p.srv.push(p.cli.output()); if (n == 0) break; p.cli.wrote(n); moved = true; } while (p.srv.retry()) |req| { p.answer(req); moved = true; } while (p.srv.next()) |req| { p.answer(req); moved = true; } while (p.srv.output().len != 0) { const n = p.cli.push(p.srv.output()); if (n == 0) break; p.srv.wrote(n); moved = true; } } } /// Submit one request, run the wire, and collect the one answer it /// produced. The tag and the op are checked here so that no test below has /// to repeat it. fn one(p: *Pair, req: Client.Request) !Client.Done { const tag = try p.cli.submit(req); p.wire(); const done = p.cli.take() orelse return error.NoReply; try testing.expectEqual(tag, done.tag); try testing.expectEqual(std.meta.activeTag(req), done.op); // One request, one reply, and nothing left outstanding: the invariant // that makes `pending()` usable as a loop condition. try testing.expectEqual(@as(usize, 0), p.cli.pending()); return done; } /// `Tversion` and `Tattach`, leaving the root on fid 0 — the client-side /// twin of `Harness.handshake`. fn handshake(p: *Pair) !void { p.start(); const v = try p.one(.{ .version = .{} }); try testing.expectEqualStrings("9P2000", v.result.version.version); try testing.expectEqual(@as(u16, notag), v.tag); const a = try p.one(.{ .attach = .{ .fid = 0, .uname = "goblin" } }); try testing.expectEqual(@as(u64, 1), a.result.attach.path); try testing.expectEqual(qtdir, a.result.attach.type); } }; test "9p client: a whole session against the server in this file" { var p: Pair = .{}; try p.handshake(); // Both ends agreed the same number, and it came off the buffers rather // than out of the air. try testing.expectEqual(@as(u32, 4096), p.cli.msize); try testing.expectEqual(@as(u32, 4096 - 11), p.cli.maxRead()); try testing.expectEqual(@as(u32, 4096 - 23), p.cli.maxWrite()); // A two-element walk onto pane 1's `body`. `nwqid` equals what was asked, // which is the only thing that says the fid landed where we wanted. const w = try p.one(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{ "1", "body" } } }); try testing.expectEqual(@as(u16, 2), w.result.walk.nwqid); try testing.expectEqual(@as(u64, 16), w.result.walk.wqid[0].path); try testing.expectEqual(@as(u64, 18), w.result.walk.wqid[1].path); try testing.expectEqual(qtfile, w.result.walk.wqid[1].type); const o = try p.one(.{ .open = .{ .fid = 1, .mode = ordwr } }); try testing.expectEqual(@as(u64, 18), o.result.open.qid.path); try testing.expectEqual(@as(u32, 4096 - iohdrsz), o.result.open.iounit); const r = try p.one(.{ .read = .{ .fid = 1, .offset = 0, .count = 64 } }); try testing.expectEqualStrings("hello, body\n", r.result.read); // Past the end is zero bytes and not an error: 9P has no EOF flag, and a // short read is how a client learns it is done. const eof = try p.one(.{ .read = .{ .fid = 1, .offset = 12, .count = 64 } }); try testing.expectEqual(@as(usize, 0), eof.result.read.len); const wr = try p.one(.{ .write = .{ .fid = 1, .offset = 0, .data = "abc" } }); try testing.expectEqual(@as(u32, 3), wr.result.write); try testing.expectEqualStrings("abc", p.fsys.writes[0..p.fsys.writes_len]); const st = try p.one(.{ .stat = .{ .fid = 1 } }); try testing.expectEqualStrings("body", st.result.stat.name); try testing.expectEqual(@as(u64, 12), st.result.stat.length); try testing.expectEqualStrings("goblin", st.result.stat.uid); _ = try p.one(.{ .clunk = .{ .fid = 1 } }); // The clunk paid the core its release, which is the half of a clunk a // client cannot see and the server tests pin down from the other side. try testing.expectEqual(@as(u32, 1), p.fsys.releases); // Nothing outstanding, nothing buffered, nothing owed. try testing.expectEqual(@as(usize, 0), p.cli.pending()); try testing.expectEqual(@as(usize, 0), p.cli.output().len); try testing.expect(p.cli.take() == null); try testing.expect(!p.cli.dead); } /// Move the server's queued replies to the client LAST FIRST. 9P permits it — /// nothing in the protocol orders replies against each other — and both /// reference clients allocate a tag per outstanding request with no in-order /// assumption anywhere (`linux/net/9p/client.c:194-199`, /// `plan9/devmnt.c:783-800`). A client that quietly relies on order works /// until the day the server answers a cached stat before a blocked read. fn deliverReversed(p: *Pair) !void { var scratch: [4096]u8 = undefined; const out = p.srv.output(); try testing.expect(out.len <= scratch.len); @memcpy(scratch[0..out.len], out); const total = out.len; p.srv.wrote(total); var at: [max_tags]usize = undefined; var lens: [max_tags]u32 = undefined; var count: usize = 0; var i: usize = 0; while (i < total) { const len = frameLen(scratch[i..total]) orelse return error.ShortReply; at[count] = i; lens[count] = len; count += 1; i += len; } try testing.expect(count >= 2); var k = count; while (k > 0) { k -= 1; const f = scratch[at[k]..][0..lens[k]]; try testing.expectEqual(f.len, p.cli.push(f)); } } test "9p client: replies out of order are matched by tag and not by arrival" { var p: Pair = .{}; try p.handshake(); const w = try p.one(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"index"} } }); try testing.expectEqual(@as(u16, 1), w.result.walk.nwqid); // Two stats outstanding at once, on two different files. const root_tag = try p.cli.submit(.{ .stat = .{ .fid = 0 } }); const index_tag = try p.cli.submit(.{ .stat = .{ .fid = 1 } }); try testing.expectEqual(@as(u16, 0), root_tag); try testing.expectEqual(@as(u16, 1), index_tag); try testing.expectEqual(@as(usize, 2), p.cli.pending()); // Both requests to the server, both replies produced, then handed back in // the wrong order. while (p.cli.output().len != 0) { const n = p.srv.push(p.cli.output()); p.cli.wrote(n); } while (p.srv.next()) |req| p.answer(req); try deliverReversed(&p); // The SECOND request answers first, and it is recognised by its tag. const first = p.cli.take() orelse return error.NoReply; try testing.expectEqual(index_tag, first.tag); try testing.expectEqualStrings("index", first.result.stat.name); const second = p.cli.take() orelse return error.NoReply; try testing.expectEqual(root_tag, second.tag); try testing.expectEqualStrings("/", second.result.stat.name); try testing.expectEqual(@as(usize, 0), p.cli.pending()); try testing.expect(!p.cli.dead); } test "9p client: an Rerror answers one operation and the session carries on" { var p: Pair = .{}; try p.handshake(); // A walk failing on its FIRST element is an `Rerror` rather than a short // `Rwalk` — the one asymmetry in walk(5), and the reason `Done` reports // the op beside the string. const bad = try p.one(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"nope"} } }); try testing.expectEqual(Client.Op.walk, bad.op); try testing.expectEqualStrings(errString(2), bad.result.fail); // The failed tag is free again and the connection is untouched: an error // is an answer, not a fault. try testing.expectEqual(@as(usize, 0), p.cli.pending()); try testing.expect(!p.cli.dead); const st = try p.one(.{ .stat = .{ .fid = 0 } }); try testing.expectEqualStrings("/", st.result.stat.name); // A refusal that comes from the server's own table rather than the core's, // spelled the way Linux's error table holds it. const stale = try p.one(.{ .stat = .{ .fid = 9 } }); try testing.expectEqualStrings(e_unknown_fid, stale.result.fail); try testing.expect(!p.cli.dead); } test "9p client: a reply arriving a byte at a time is taken when its last byte lands" { var p: Pair = .{}; try p.handshake(); const tag = try p.cli.submit(.{ .stat = .{ .fid = 0 } }); // The request out, the reply produced, and then held on this side of the // wire so it can be dripped in. while (p.cli.output().len != 0) { const n = p.srv.push(p.cli.output()); p.cli.wrote(n); } while (p.srv.next()) |req| p.answer(req); var scratch: [512]u8 = undefined; const out = p.srv.output(); try testing.expect(out.len > 4 and out.len <= scratch.len); @memcpy(scratch[0..out.len], out); const reply = scratch[0..out.len]; p.srv.wrote(reply.len); // Every byte but the last leaves nothing to collect — including the first // four, where `frameLen` becomes readable and still says "wait". for (reply[0 .. reply.len - 1]) |b| { try testing.expectEqual(@as(usize, 1), p.cli.push(&.{b})); try testing.expect(p.cli.take() == null); try testing.expect(!p.cli.dead); } try testing.expectEqual(@as(usize, 1), p.cli.push(reply[reply.len - 1 ..])); const done = p.cli.take() orelse return error.NoReply; try testing.expectEqual(tag, done.tag); try testing.expectEqualStrings("/", done.result.stat.name); } test "9p client: sixteen tags outstanding, and the seventeenth is refused" { var p: Pair = .{}; try p.handshake(); // Nothing is wired, so nothing is answered and every tag stays out. var tags: [max_tags]u16 = undefined; for (&tags, 0..) |*t, i| { t.* = try p.cli.submit(.{ .stat = .{ .fid = 0 } }); // Lowest free tag first, which is what makes a trace readable. try testing.expectEqual(@as(u16, @intCast(i)), t.*); } try testing.expectEqual(max_tags, p.cli.pending()); try testing.expectError(error.NoTags, p.cli.submit(.{ .stat = .{ .fid = 0 } })); // A refused submit costs nothing: no tag, and not a byte in the queue. const owed = p.cli.output().len; try testing.expectError(error.NoTags, p.cli.submit(.{ .clunk = .{ .fid = 0 } })); try testing.expectEqual(owed, p.cli.output().len); // Drained, every tag comes back, and the seventeenth request now fits. p.wire(); var seen: [max_tags]bool = @splat(false); for (0..max_tags) |_| { const done = p.cli.take() orelse return error.NoReply; try testing.expectEqual(Client.Op.stat, done.op); try testing.expect(!seen[done.tag]); seen[done.tag] = true; } for (seen) |s| try testing.expect(s); try testing.expectEqual(@as(usize, 0), p.cli.pending()); _ = try p.one(.{ .stat = .{ .fid = 0 } }); } test "9p client: what a caller may not ask for is refused before a tag is spent" { var p: Pair = .{}; p.start(); // Nothing before the handshake, and `Tversion` is the only exception. try testing.expectError(error.Handshake, p.cli.submit(.{ .stat = .{ .fid = 0 } })); try p.handshake(); // `NOFID` is not a fid a client may name. try testing.expectError(error.BadRequest, p.cli.submit(.{ .stat = .{ .fid = nofid } })); try testing.expectError(error.BadRequest, p.cli.submit(.{ .clunk = .{ .fid = nofid } })); try testing.expectError(error.BadRequest, p.cli.submit(.{ .walk = .{ .fid = 0, .newfid = nofid, .names = &.{} } })); // A path that has not been split, and one longer than the protocol admits. try testing.expectError(error.BadRequest, p.cli.submit(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{"1/body"} } })); try testing.expectError(error.BadRequest, p.cli.submit(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{""} } })); const seventeen: [max_welem + 1][]const u8 = @splat("x"); try testing.expectError(error.BadRequest, p.cli.submit(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &seventeen } })); // Both I/O bounds, each one byte past what the msize can carry. try testing.expectError(error.TooLarge, p.cli.submit(.{ .read = .{ .fid = 0, .offset = 0, .count = p.cli.maxRead() + 1, } })); var big: [4096]u8 = @splat('x'); try testing.expectError(error.TooLarge, p.cli.submit(.{ .write = .{ .fid = 0, .offset = 0, .data = big[0 .. p.cli.maxWrite() + 1], } })); // And exactly at the bound, both fit — a cap that is off by one is a cap // that costs a round trip on every large transfer. Each one on an EMPTY // queue, which is what `Options.out` means by "room for one whole // request": a maximum-size `Twrite` IS the msize, so it fits beside // nothing at all. p.cli.wrote(p.cli.output().len); _ = try p.cli.submit(.{ .read = .{ .fid = 0, .offset = 0, .count = p.cli.maxRead() } }); p.cli.wrote(p.cli.output().len); _ = try p.cli.submit(.{ .write = .{ .fid = 0, .offset = 0, .data = big[0..p.cli.maxWrite()] } }); // A second `Tversion` resets the connection, so it is refused while // anything is outstanding rather than abandoning it. try testing.expectError(error.Handshake, p.cli.submit(.{ .version = .{} })); // And with the queue full of that one write there is nowhere to put even // an eleven-byte `Tstat`: the out queue is the only back-pressure a // sans-io client has, and it lands at `submit` where the caller is // standing right there. try testing.expectError(error.NoSpace, p.cli.submit(.{ .stat = .{ .fid = 0 } })); } test "9p client: an msize below the floor, and one the server tried to raise" { var in: [512]u8 = undefined; var out: [512]u8 = undefined; var buf: [64]u8 = undefined; // A caller asking for less than an `Rwalk` is asking for a connection that // cannot be served. var c = Client.init(.{ .in = &in, .out = &out }); try testing.expectError(error.BadRequest, c.submit(.{ .version = .{ .msize = msize_min - 1 } })); // Zero means "whatever the buffers hold", which is the smaller of the two. _ = try c.submit(.{ .version = .{} }); try testing.expectEqual(@as(u32, 512), c.asked); // A server answering with MORE than the client offered is a server whose // next message will not fit the buffer that has to hold it. _ = c.push(try encode(.{ .rversion = .{ .msize = 1024, .version = "9P2000" } }, notag, &buf)); try testing.expect(c.take() == null); try testing.expect(c.dead); // "unknown" is a SUCCESSFUL reply with no dialect in common: the handshake // completes, `msize` stays zero, and nothing more can be submitted. var c2 = Client.init(.{ .in = &in, .out = &out }); _ = try c2.submit(.{ .version = .{} }); _ = c2.push(try encode(.{ .rversion = .{ .msize = 512, .version = "unknown" } }, notag, &buf)); const done = c2.take() orelse return error.NoReply; try testing.expectEqualStrings("unknown", done.result.version.version); try testing.expect(!c2.dead); try testing.expectEqual(@as(u32, 0), c2.msize); try testing.expectError(error.Handshake, c2.submit(.{ .stat = .{ .fid = 0 } })); // A dialect we never offered is neither: agreeing to it would be agreeing // to a layout this file does not implement. var c3 = Client.init(.{ .in = &in, .out = &out }); _ = try c3.submit(.{ .version = .{} }); _ = c3.push(try encode(.{ .rversion = .{ .msize = 512, .version = "9P2000.u" } }, notag, &buf)); try testing.expect(c3.take() == null); try testing.expect(c3.dead); } test "9p client: what is not an answer to one of our requests ends the connection" { var buf: [64]u8 = undefined; // Each case gets a fresh connection past the handshake, because every one // of them is fatal by design. const Case = struct { fn armed(in: []u8, out: []u8, scratch: []u8) !Client { var c = Client.init(.{ .in = in, .out = out }); _ = try c.submit(.{ .version = .{} }); c.wrote(c.output().len); _ = c.push(try encode(.{ .rversion = .{ .msize = 512, .version = "9P2000" } }, notag, scratch)); _ = c.take() orelse return error.NoReply; _ = try c.submit(.{ .stat = .{ .fid = 0 } }); c.wrote(c.output().len); return c; } }; var in: [512]u8 = undefined; var out: [512]u8 = undefined; // A T-message. A client reads R-messages, and the parity says so with no // table: this is `Server.startFrame`'s refusal read from the other end. { var c = try Case.armed(&in, &out, &buf); _ = c.push(try encode(.{ .tstat = .{ .fid = 0 } }, 0, &buf)); try testing.expect(c.take() == null); try testing.expect(c.dead); } // A reply on a tag nobody claimed. { var c = try Case.armed(&in, &out, &buf); _ = c.push(try encode(.rclunk, 3, &buf)); try testing.expect(c.take() == null); try testing.expect(c.dead); } // A tag outside the table entirely, which no reply to us can carry. { var c = try Case.armed(&in, &out, &buf); _ = c.push(try encode(.rclunk, 900, &buf)); try testing.expect(c.take() == null); try testing.expect(c.dead); } // The right tag and the WRONG SHAPE: an `Rclunk` where an `Rstat` was // asked for. A tag is only as good as the table behind it. { var c = try Case.armed(&in, &out, &buf); _ = c.push(try encode(.rclunk, 0, &buf)); try testing.expect(c.take() == null); try testing.expect(c.dead); } // A reply to a request this client never sends. { var c = try Case.armed(&in, &out, &buf); _ = c.push(try encode(.rwstat, 0, &buf)); try testing.expect(c.take() == null); try testing.expect(c.dead); } // A `size` no encoder produced, and one past the negotiated msize. Both // are streams that will never resynchronise. { var c = try Case.armed(&in, &out, &buf); _ = c.push(&.{ 3, 0, 0, 0 }); try testing.expect(c.take() == null); try testing.expect(c.dead); } { var c = try Case.armed(&in, &out, &buf); _ = c.push(&.{ 0, 4, 0, 0 }); try testing.expect(c.take() == null); try testing.expect(c.dead); } // And a body the codec refuses: the type byte is fine, the payload is not. { var c = try Case.armed(&in, &out, &buf); _ = c.push(&.{ 8, 0, 0, 0, @intFromEnum(Type.rstat), 0, 0, 0 }); try testing.expect(c.take() == null); try testing.expect(c.dead); } } test "9p client: an Rread longer than the Tread asked for is refused" { // The one bound a client cannot check from the frame alone, which is why // `Slot` keeps the count: a server handing back more than was asked has // given us bytes at offsets we never named. Linux calls it `-EIO`. var p: Pair = .{}; try p.handshake(); _ = try p.one(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{ "1", "body" } } }); _ = try p.one(.{ .open = .{ .fid = 1, .mode = oread } }); const tag = try p.cli.submit(.{ .read = .{ .fid = 1, .offset = 0, .count = 4 } }); p.cli.wrote(p.cli.output().len); var buf: [64]u8 = undefined; _ = p.cli.push(try encode(.{ .rread = .{ .data = "hello, body\n" } }, tag, &buf)); try testing.expect(p.cli.take() == null); try testing.expect(p.cli.dead); // Exactly the count asked for is fine, and so is anything shorter. var q: Pair = .{}; try q.handshake(); _ = try q.one(.{ .walk = .{ .fid = 0, .newfid = 1, .names = &.{ "1", "body" } } }); _ = try q.one(.{ .open = .{ .fid = 1, .mode = oread } }); const short = try q.one(.{ .read = .{ .fid = 1, .offset = 0, .count = 4 } }); try testing.expectEqualStrings("hell", short.result.read); } test "9p client: hangup and a dead connection refuse everything after" { var p: Pair = .{}; try p.handshake(); p.cli.hangup(); try testing.expectEqual(@as(usize, 0), p.cli.pending()); try testing.expectEqual(@as(usize, 0), p.cli.output().len); try testing.expectEqual(@as(usize, 0), p.cli.push("anything")); try testing.expect(p.cli.take() == null); try testing.expectError(error.Dead, p.cli.submit(.{ .stat = .{ .fid = 0 } })); try testing.expectError(error.Dead, p.cli.submit(.{ .version = .{} })); } test "9p client: one session is a hundred and change bytes plus its buffers" { // The number the board is costed against, and the whole reason the client // is shaped the way it is: sixteen eight-byte tag slots and nine scalars, // with every payload borrowed out of the input buffer rather than copied // into a per-tag one. A `Server` on the same connection is 9,488 B because // it owns a fid table and a park table; a client owns neither, because the // far end does. try testing.expect(@sizeOf(Client.Slot) <= 8); try testing.expect(@sizeOf(Client) <= 256); // Two buffers at the 8,192-byte msize `src/fs9_service.zig` serves, plus // the client itself: what one `9p` word costs while it is running. try testing.expect(2 * 8192 + @sizeOf(Client) <= 17 * 1024); // And at the protocol floor, which is what a board would negotiate: two // buffers of 217 bytes each is a 9P client in under 700 bytes of RAM. try testing.expect(2 * msize_min + @sizeOf(Client) <= 700); }