//! THE MACHINE-LOCAL HALF OF A HOST: fork a pane's shell, put bytes on a disk. //! //! `host.zig` is the seam — the struct of function pointers the core asks //! through. This file is the part of the answer that is the same on every host //! that has an operating system under it, and it is now the ONLY copy of it: //! tty.zig, detached/server.zig, gui/gui.zig and macos.zig all fork and write //! through here. They did not always. Each of the four grew its own `forkShell` //! and its own `writeFd`, and what those four copies were for is best said by //! what they had in common: ALL FOUR were missing FD_CLOEXEC on the pty master, //! so in every shell pardes has ever shipped a program in one pane could read //! and write another pane's terminal, and closing a master did not reliably hang //! its shell up. One line below fixes that for all four at once (see `forkShell`) //! — which is a better argument for this file existing than "it is shared" is. //! //! Why the daemon and not the frontend does this work: a unix socket means the //! core and its frontends are on the SAME machine, so there is no question of //! whose disk or whose process table is meant. Given that, the pane shells //! belong to the long-lived process, because the whole promise of a detached //! session is that it outlives the frontend attached to it — a shell forked by //! a frontend dies with that frontend, and then the session has a pane with no //! shell in it. The frontend keeps exactly what needs the human's screen: the //! grid, the keyboard, the clipboard and a link to open. //! //! So `forkShell` takes the core it is forking on behalf of and nothing about //! terminals: no vaxis, no `Loop`, no reader thread. Who drains the master fd //! is the caller's business, and the callers answer differently on purpose. The //! tty, gui and macOS shells hand it to a worker that posts into their event //! loop; the daemon adds it to the one `poll(2)` it already runs over its //! clients, and makes its own copy non-blocking in order to. That last is why //! `Child.file.flags` is left saying what it says: the flag describes the //! descriptor `forkpty` handed back, for the three callers that stream it, and //! the one that polls it keeps only the handle. const std = @import("std"); const posix = std.posix; const libc = std.c; const pardes = @import("pardes.zig"); const shell_bin = @import("shell_bin.zig"); const fs_service = @import("fs_service.zig"); const fuse = @import("fuse.zig"); /// `setCloexec` and nothing else. Imported rather than copied a fourth time — /// fuse.zig and nested.zig each grew a private two-line version of it — because /// the descriptor this file has to protect is the one every OTHER file in the /// tree already protects, and one predicate is how the reasoning stays in one /// place. nested.zig is a leaf (std, builtin, libc), so this costs no /// dependency worth the name. const nested = @import("nested.zig"); extern "c" fn forkpty(amaster: *c_int, name: ?[*:0]u8, termp: ?*const anyopaque, winp: ?*const posix.winsize) c_int; extern "c" fn execv(path: [*:0]const u8, argv: [*:null]const ?[*:0]const u8) c_int; extern "c" fn chdir(path: [*:0]const u8) c_int; extern "c" fn _exit(status: c_int) noreturn; /// A forked pane shell: the pty master to read and write, and the pid to reap. /// Named rather than anonymous because four files now hold one of these. pub const Child = struct { file: std.Io.File, pid: posix.pid_t, }; /// Fork a shell onto a fresh pty for `pane`, sized `rows`x`cols`. /// /// `core` is optional because a host may fork before it has one, and a core /// that is absent simply does not name its shell. pub fn forkShell( core: ?*pardes.Pardes, pane: usize, prompt_rcs: *const shell_bin.PromptRcs, bin: []const u8, cwd: ?[*:0]const u8, rows: u16, cols: u16, fs: ?*const fuse.Fs, ) Child { var master: c_int = undefined; // resolved BEFORE the fork, into this frame, which the child inherits: // nothing between fork and exec may allocate, and a PATH search would var path_buf: [std.fs.max_path_bytes]u8 = undefined; const spawn = shell_bin.resolve(bin, &path_buf, prompt_rcs); // ...and so is the pane's own address on the control filesystem, for a // second reason on top of that one: acme puts `winid` in the child, which // is safe there only because rfork(RFENVG) has just given it a private // environment group. See fs_service.exportPaneEnv. fs_service.exportPaneEnv(fs, if (core) |c| (if (c.panes[pane]) |pn| pn.serial else 0) else 0); const ws = posix.winsize{ .row = rows, .col = cols, .xpixel = 0, .ypixel = 0 }; const pid = forkpty(&master, null, null, &ws); if (pid == 0) { // the blocked-SIGWINCH mask survives fork AND exec — unblock it or // bash/vim in the pane would never see resizes (sigprocmask is // async-signal-safe) var set = posix.sigemptyset(); posix.sigaddset(&set, posix.SIG.WINCH); posix.sigprocmask(posix.SIG.UNBLOCK, &set, null); if (cwd) |c| _ = chdir(c); _ = execv(spawn.path, &spawn.argv); _exit(127); } if (pid > 0) { // CLOEXEC ON THE MASTER, and it belongs here rather than at either // caller because `forkpty` is what opens it: /dev/ptmx is opened with no // O_CLOEXEC and there is no flag argument to ask for one. Without this, // every pane shell forked AFTER this one inherits this master and keeps // it across `execv`, which is two bugs at once. // // The loud one: a program running in pane 3 can read pane 0's output and // write bytes into pane 0's screen. // // The silent one, and the reason it compounds: closing a master is the // only thing that hangs its shell up, and a master a later shell still // holds open is not closed. detached/server.zig `closePty` and tty.zig // `spawn` both depend on that hangup, so a pane delete or a respawn left // an orphaned shell that never exits — never reaped, eventually blocked // writing into a pty nobody reads — and each orphan pinned every earlier // pane's master in turn. The startup drain forks pane 0 and then pane 1, // so the arrangement existed from boot, and it existed in all four // copies of this function before they became this one. nested.zig and // fuse.zig say the same thing about their own descriptors ("pane shells // are forked with forkpty and inherit everything open"); the master was // the one descriptor in the tree that nobody had said it to. // // THE WINDOW THIS LEAVES, stated rather than papered over: fcntl after // fork is not atomic, so a thread that forks and execs between these two // syscalls inherits the master anyway. In the detached daemon there is no // such thread — it is single-threaded by construction, which is what // putting the pty masters in its own `poll(2)` bought. The shells with // worker threads that can exec — tty.zig's pipe tasks above all — have a // window two syscalls wide, and closing it means replacing `forkpty` with // our own `posix_openpt(O_CLOEXEC)` / `grantpt` / `unlockpt` / fork / // `setsid`, which is a different change to a different file. nested.setCloexec(master); if (core) |c| c.acknowledgeShell(pane, std.mem.span(spawn.path), spawn.argv[1] != null); } return .{ .file = .{ .handle = master, .flags = .{ .nonblocking = false } }, .pid = pid }; } /// Truncate-or-create `path` and put `bytes` there. False on any failure, and /// the caller reports it: a save that did not happen must not be announced as /// one. pub fn writeFileBytes(path: []const u8, bytes: []const u8) bool { var pathbuf: [4096:0]u8 = undefined; if (path.len >= pathbuf.len) return false; @memcpy(pathbuf[0..path.len], path); pathbuf[path.len] = 0; const fd = libc.open(pathbuf[0..path.len :0], .{ .ACCMODE = .WRONLY, .CREAT = true, .TRUNC = true }, @as(libc.mode_t, 0o644)); if (fd < 0) return false; writeFd(fd, bytes); _ = libc.close(fd); return true; } /// A whole-buffer write that finishes short writes, retries EINTR, and refuses /// to loop on no progress. /// /// The zero guard is not bookkeeping: without it a `write(2)` that returns 0 for /// a nonzero count is an infinite SPIN, because 0 is neither an error nor /// progress and `off` never moves. macos.zig's copy carried the guard and its /// reason all along — "a zero-byte write makes no progress; looping on it would /// spin the main thread forever" — and the tty copy this file was extracted /// from did not, so the extraction briefly promoted the weakest of the three to /// being the shared one. All three are now this one: gui.zig and macos.zig were /// migrated onto it, so the guard is no longer missing anywhere. /// /// A spin is strictly worse than the block it replaces, which is why this /// matters more now that detached/server.zig reaches this file from a /// single-threaded poll loop: a blocked `write` is one syscall a signal can /// interrupt, and a spin is 100% of a core with the whole session behind it. pub fn writeFd(fd: c_int, data: []const u8) void { var off: usize = 0; while (off < data.len) { const n = libc.write(fd, data[off..].ptr, data.len - off); if (n < 0) { if (libc.errno(n) == .INTR) continue; return; } if (n == 0) return; off += @intCast(n); } }