summaryrefslogtreecommitdiff
path: root/src/host_io.zig
blob: 6ffc890eb3c95efef2b44f8968414ab1f5a9ca3c (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
//! THE MACHINE-LOCAL HALF OF A HOST: fork a pane's shell, put bytes on a disk.
//!
//! `host.zig` is the seam — the struct of function pointers the core asks
//! through. This file is the part of the answer that is the same on every host
//! that has an operating system under it, and it is now the ONLY copy of it:
//! tty.zig, detached/server.zig, gui/gui.zig and macos.zig all fork and write
//! through here. They did not always. Each of the four grew its own `forkShell`
//! and its own `writeFd`, and what those four copies were for is best said by
//! what they had in common: ALL FOUR were missing FD_CLOEXEC on the pty master,
//! so in every shell pardes has ever shipped a program in one pane could read
//! and write another pane's terminal, and closing a master did not reliably hang
//! its shell up. One line below fixes that for all four at once (see `forkShell`)
//! — which is a better argument for this file existing than "it is shared" is.
//!
//! Why the daemon and not the frontend does this work: a unix socket means the
//! core and its frontends are on the SAME machine, so there is no question of
//! whose disk or whose process table is meant. Given that, the pane shells
//! belong to the long-lived process, because the whole promise of a detached
//! session is that it outlives the frontend attached to it — a shell forked by
//! a frontend dies with that frontend, and then the session has a pane with no
//! shell in it. The frontend keeps exactly what needs the human's screen: the
//! grid, the keyboard, the clipboard and a link to open.
//!
//! So `forkShell` takes the core it is forking on behalf of and nothing about
//! terminals: no vaxis, no `Loop`, no reader thread. Who drains the master fd
//! is the caller's business, and the callers answer differently on purpose. The
//! tty, gui and macOS shells hand it to a worker that posts into their event
//! loop; the daemon adds it to the one `poll(2)` it already runs over its
//! clients, and makes its own copy non-blocking in order to. That last is why
//! `Child.file.flags` is left saying what it says: the flag describes the
//! descriptor `forkpty` handed back, for the three callers that stream it, and
//! the one that polls it keeps only the handle.
const std = @import("std");
const posix = std.posix;
const libc = std.c;
const pardes = @import("pardes.zig");
const shell_bin = @import("shell_bin.zig");
const fs_service = @import("fs_service.zig");
const fuse = @import("fuse.zig");

/// `setCloexec` and nothing else. Imported rather than copied a fourth time —
/// fuse.zig and nested.zig each grew a private two-line version of it — because
/// the descriptor this file has to protect is the one every OTHER file in the
/// tree already protects, and one predicate is how the reasoning stays in one
/// place. nested.zig is a leaf (std, builtin, libc), so this costs no
/// dependency worth the name.
const nested = @import("nested.zig");

extern "c" fn forkpty(amaster: *c_int, name: ?[*:0]u8, termp: ?*const anyopaque, winp: ?*const posix.winsize) c_int;
extern "c" fn execv(path: [*:0]const u8, argv: [*:null]const ?[*:0]const u8) c_int;
extern "c" fn chdir(path: [*:0]const u8) c_int;
extern "c" fn _exit(status: c_int) noreturn;

/// A forked pane shell: the pty master to read and write, and the pid to reap.
/// Named rather than anonymous because four files now hold one of these.
pub const Child = struct {
    file: std.Io.File,
    pid: posix.pid_t,
};

/// Fork a shell onto a fresh pty for `pane`, sized `rows`x`cols`.
///
/// `core` is optional because a host may fork before it has one, and a core
/// that is absent simply does not name its shell.
pub fn forkShell(
    core: ?*pardes.Pardes,
    pane: usize,
    prompt_rcs: *const shell_bin.PromptRcs,
    bin: []const u8,
    cwd: ?[*:0]const u8,
    rows: u16,
    cols: u16,
    fs: ?*const fuse.Fs,
) Child {
    var master: c_int = undefined;
    // resolved BEFORE the fork, into this frame, which the child inherits:
    // nothing between fork and exec may allocate, and a PATH search would
    var path_buf: [std.fs.max_path_bytes]u8 = undefined;
    const spawn = shell_bin.resolve(bin, &path_buf, prompt_rcs);
    // ...and so is the pane's own address on the control filesystem, for a
    // second reason on top of that one: acme puts `winid` in the child, which
    // is safe there only because rfork(RFENVG) has just given it a private
    // environment group. See fs_service.exportPaneEnv.
    fs_service.exportPaneEnv(fs, if (core) |c| (if (c.panes[pane]) |pn| pn.serial else 0) else 0);
    const ws = posix.winsize{ .row = rows, .col = cols, .xpixel = 0, .ypixel = 0 };
    const pid = forkpty(&master, null, null, &ws);
    if (pid == 0) {
        // the blocked-SIGWINCH mask survives fork AND exec — unblock it or
        // bash/vim in the pane would never see resizes (sigprocmask is
        // async-signal-safe)
        var set = posix.sigemptyset();
        posix.sigaddset(&set, posix.SIG.WINCH);
        posix.sigprocmask(posix.SIG.UNBLOCK, &set, null);
        if (cwd) |c| _ = chdir(c);
        _ = execv(spawn.path, &spawn.argv);
        _exit(127);
    }
    if (pid > 0) {
        // CLOEXEC ON THE MASTER, and it belongs here rather than at either
        // caller because `forkpty` is what opens it: /dev/ptmx is opened with no
        // O_CLOEXEC and there is no flag argument to ask for one. Without this,
        // every pane shell forked AFTER this one inherits this master and keeps
        // it across `execv`, which is two bugs at once.
        //
        // The loud one: a program running in pane 3 can read pane 0's output and
        // write bytes into pane 0's screen.
        //
        // The silent one, and the reason it compounds: closing a master is the
        // only thing that hangs its shell up, and a master a later shell still
        // holds open is not closed. detached/server.zig `closePty` and tty.zig
        // `spawn` both depend on that hangup, so a pane delete or a respawn left
        // an orphaned shell that never exits — never reaped, eventually blocked
        // writing into a pty nobody reads — and each orphan pinned every earlier
        // pane's master in turn. The startup drain forks pane 0 and then pane 1,
        // so the arrangement existed from boot, and it existed in all four
        // copies of this function before they became this one. nested.zig and
        // fuse.zig say the same thing about their own descriptors ("pane shells
        // are forked with forkpty and inherit everything open"); the master was
        // the one descriptor in the tree that nobody had said it to.
        //
        // THE WINDOW THIS LEAVES, stated rather than papered over: fcntl after
        // fork is not atomic, so a thread that forks and execs between these two
        // syscalls inherits the master anyway. In the detached daemon there is no
        // such thread — it is single-threaded by construction, which is what
        // putting the pty masters in its own `poll(2)` bought. The shells with
        // worker threads that can exec — tty.zig's pipe tasks above all — have a
        // window two syscalls wide, and closing it means replacing `forkpty` with
        // our own `posix_openpt(O_CLOEXEC)` / `grantpt` / `unlockpt` / fork /
        // `setsid`, which is a different change to a different file.
        nested.setCloexec(master);
        if (core) |c| c.acknowledgeShell(pane, std.mem.span(spawn.path), spawn.argv[1] != null);
    }
    return .{ .file = .{ .handle = master, .flags = .{ .nonblocking = false } }, .pid = pid };
}

/// Truncate-or-create `path` and put `bytes` there. False on any failure, and
/// the caller reports it: a save that did not happen must not be announced as
/// one.
/// WHY it failed, and not merely that it did. A save is the one operation in
/// this program whose failure a user must not be able to miss, and until this
/// returned an error there was nothing for a host to put on the message row:
/// the bool said "no" and every caller answered it with a bare `return`.
/// `NoSpaceLeft` is the one that most needs saying — the file has already been
/// truncated by the time it happens, so a save that reports nothing has
/// destroyed the file it was asked to preserve.
pub const WriteError = error{
    PathTooLong,
    PermissionDenied,
    IsDirectory,
    ReadOnlyFilesystem,
    NoSpaceLeft,
    OpenFailed,
    WriteFailed,
};

pub fn writeFileBytes(path: []const u8, bytes: []const u8) WriteError!void {
    var pathbuf: [4096:0]u8 = undefined;
    if (path.len >= pathbuf.len) return error.PathTooLong;
    @memcpy(pathbuf[0..path.len], path);
    pathbuf[path.len] = 0;
    const fd = libc.open(pathbuf[0..path.len :0], .{ .ACCMODE = .WRONLY, .CREAT = true, .TRUNC = true }, @as(libc.mode_t, 0o644));
    if (fd < 0) return switch (libc.errno(fd)) {
        .ACCES, .PERM => error.PermissionDenied,
        .ISDIR => error.IsDirectory,
        .ROFS => error.ReadOnlyFilesystem,
        .NOSPC, .DQUOT => error.NoSpaceLeft,
        .NAMETOOLONG => error.PathTooLong,
        else => error.OpenFailed,
    };
    const wrote = writeFd(fd, bytes);
    // The close is part of the write. NFS and every write-back filesystem
    // report a deferred error here and nowhere else, so a close that fails on a
    // file we believe we wrote is a file we did not write.
    const closed = libc.close(fd) == 0;
    if (!wrote or !closed) return error.WriteFailed;
}

/// A whole-buffer write that finishes short writes, retries EINTR, and refuses
/// to loop on no progress.
///
/// The zero guard is not bookkeeping: without it a `write(2)` that returns 0 for
/// a nonzero count is an infinite SPIN, because 0 is neither an error nor
/// progress and `off` never moves. macos.zig's copy carried the guard and its
/// reason all along — "a zero-byte write makes no progress; looping on it would
/// spin the main thread forever" — and the tty copy this file was extracted
/// from did not, so the extraction briefly promoted the weakest of the three to
/// being the shared one. All three are now this one: gui.zig and macos.zig were
/// migrated onto it, so the guard is no longer missing anywhere.
///
/// A spin is strictly worse than the block it replaces, which is why this
/// matters more now that detached/server.zig reaches this file from a
/// single-threaded poll loop: a blocked `write` is one syscall a signal can
/// interrupt, and a spin is 100% of a core with the whole session behind it.
/// True when every byte went. The answer is new: this used to return `void`, so
/// a full disk and a completed write were the same event to every caller — and
/// the one caller that matters had already truncated the file. A pty write
/// ignores it, which is what `_ =` at those call sites means.
pub fn writeFd(fd: c_int, data: []const u8) bool {
    var off: usize = 0;
    while (off < data.len) {
        const n = libc.write(fd, data[off..].ptr, data.len - off);
        if (n < 0) {
            if (libc.errno(n) == .INTR) continue;
            return false;
        }
        if (n == 0) return false;
        off += @intCast(n);
    }
    return true;
}