1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
|
//! The crash file: what a panic leaves behind for the human who has to report it.
//!
//! stderr is where a panic has always gone, and stderr is the worst place this
//! program has. In the tty shell stderr IS the screen — the trace lands on the
//! grid the terminal is being reset out of, and the next redraw or the next
//! `clear` takes it with it. The GUI and the AppKit shells have no terminal at
//! all, so it goes to a console nobody has open. A detached session's goes
//! wherever the launcher left it, which is usually /dev/null. So a record is
//! appended to `<config dir>/crashes` as well, under one line naming the build
//! that produced it: a bug report needs the version and the commit, and a user
//! who runs pardes already knows where its config lives.
//!
//! The record is the metadata line and the panic message, and NOT the frames:
//! every way of collecting those from out here can wedge the process instead of
//! ending it. `record` says exactly why, at length, because the reasoning is
//! not visible from the call it does not make.
//!
//! Panics only. A SIGSEGV never reaches a panic handler (main.zig `debug`
//! hooks that path separately) and unwinding one needs the signal's saved cpu
//! context, which is a different job than this one.
const std = @import("std");
const builtin = @import("builtin");
const pardes = @import("pardes.zig");
/// Beside `init`, so `Config` opens the directory that holds both.
pub const name = "crashes";
/// The config directory `config.User.load` resolved, COPIED rather than
/// borrowed: main.zig hands over an arena slice that outlives the process, but
/// the AppKit host's lives in a `config_arena` its own `errdefer` frees on a
/// failed init and its teardown frees at quit — and a panic after either would
/// have built a path out of freed memory and then created a directory at it.
/// A panic handler is the one caller that cannot check whether its input is
/// still alive, so it does not borrow.
var dir_buf: [1024]u8 = undefined;
var dir_len: usize = 0;
/// Called where the launcher sets `Options.config_dir`. Ignores a path too long
/// to hold, which is a crash file that never appears rather than a truncated
/// path pointing somewhere real.
pub fn setDir(path: []const u8) void {
if (path.len == 0 or path.len > dir_buf.len) return;
@memcpy(dir_buf[0..path.len], path);
dir_len = path.len;
}
/// Built at comptime out of the two halves main.zig's `--version` prints, and
/// for the same reason: there is nothing here to format at runtime. A build
/// from a tarball says `pardes 0.0.1` rather than inventing a revision.
const build_id = if (pardes.commit) |c|
"pardes " ++ pardes.version ++ " (" ++ c ++ ")"
else
"pardes " ++ pardes.version;
/// The whole record is formatted here before a byte of it is written, because
/// the fd is opened `O_APPEND` and one `write(2)` of the lot is what keeps two
/// crashing processes from interleaving their records line by line. Static
/// rather than a local: a panic handler may be running on a thread stack of a
/// few pages. Two lines fit several times over — the panic message is the only
/// part with no bound of its own, and a longer one is simply cut.
var scratch: [8 << 10]u8 = undefined;
/// One record per process, taken by whoever panics first. `defaultPanic` gets
/// the same property from a private `panic_stage` and from the stderr lock, and
/// neither is reachable from out here: two threads panicking at once would
/// interleave `@memcpy`s into one `scratch` and each write out the mixture, and
/// the nested panic `defaultPanic`'s own trace printer can raise comes back
/// through the handler and appends its wreckage under the real message. Both
/// end here instead, and the loser falls through to `defaultPanic`, which still
/// says everything on stderr.
var recording: std.atomic.Value(bool) = .init(false);
/// Append one crash record, or silently do nothing. Called FROM the panic
/// handler, so it takes no lock of this program's, allocates nothing of its
/// own, and does its own file I/O: a static buffer, one `open`, one `write`.
/// Every failure is swallowed — a crash file that could not be written must not
/// become the crash, and stderr is about to get the message either way.
pub fn record(msg: []const u8) void {
if (dir_len == 0) return;
// ONCE PER PROCESS, and never released. A process panics once and dies; a
// SECOND record is always the wreckage of the first. Observed with the
// flag released on the way out: `defaultPanic`'s own trace printing raised
// a nested panic, which came back through the handler and appended
// "panic: reached unreachable code" under the real message — burying the
// one line the file exists to keep. A test resets it deliberately.
if (recording.swap(true, .seq_cst)) return;
const config_dir = dir_buf[0..dir_len];
var path_buf: [1024:0]u8 = undefined;
if (config_dir.len + 1 + name.len >= path_buf.len) return;
// The directory is where the config WOULD be, not where it is: a user who
// never wrote an init file has no config directory, and the first thing
// they have to report is this. Parents are assumed and EEXIST is the
// normal answer, exactly as dump.zig bets for its own directory.
@memcpy(path_buf[0..config_dir.len], config_dir);
path_buf[config_dir.len] = 0;
_ = std.c.mkdir(path_buf[0..config_dir.len :0], 0o755);
path_buf[config_dir.len] = '/';
@memcpy(path_buf[config_dir.len + 1 ..][0..name.len], name);
path_buf[config_dir.len + 1 + name.len] = 0;
const path = path_buf[0 .. config_dir.len + 1 + name.len :0];
const fd = std.c.open(path, .{
.ACCMODE = .WRONLY,
.CREAT = true,
.APPEND = true,
}, @as(std.c.mode_t, 0o600));
if (fd < 0) return;
defer _ = std.c.close(fd);
var w: std.Io.Writer = .fixed(&scratch);
// The metadata line. Same clock and the same civil-time arithmetic as
// dump.zig's filename, printed as UTC because a crash file is read by
// whoever the reporter sends it to and their zone is not the reporter's.
//
// The return IS checked, unlike dump.zig's, because this one runs where a
// panic cannot be afforded: a failed `clock_gettime` leaves `ts` undefined,
// and an undefined large-positive `sec` walks `calculateYearDay`'s `u16`
// year past 65535 and overflow-panics INSIDE the panic handler. Debug fills
// it with 0xaa and lands in 1970, which is why it reads as harmless.
var ts: std.c.timespec = .{ .sec = 0, .nsec = 0 };
if (std.c.clock_gettime(.REALTIME, &ts) != 0) ts = .{ .sec = 0, .nsec = 0 };
const es: std.time.epoch.EpochSeconds = .{ .secs = @intCast(@max(0, ts.sec)) };
const yd = es.getEpochDay().calculateYearDay();
const md = yd.calculateMonthDay();
const ds = es.getDaySeconds();
w.print("\n" ++ build_id ++ " {d:0>4}-{d:0>2}-{d:0>2}T{d:0>2}:{d:0>2}:{d:0>2}Z {s}-{s} pid {d}\n", .{
yd.year,
md.month.numeric(),
@as(u8, md.day_index) + 1,
ds.getHoursIntoDay(),
ds.getMinutesIntoHour(),
ds.getSecondsIntoMinute(),
@tagName(builtin.os.tag),
@tagName(builtin.cpu.arch),
// Unsigned: `{d}` prints a leading '+' for a positive SIGNED int, which
// is the same note main.zig makes where it names a `--detach` session
// after this pid.
@as(u32, @intCast(std.c.getpid())),
}) catch {};
// ...and under it the line stderr is about to print.
//
// NO STACK TRACE, AND NOT EVEN THE RAW RETURN ADDRESSES. This is the whole
// design decision in this file, and it was arrived at by measurement, so it
// is written down here rather than left to be rediscovered.
//
// `std.debug.writeCurrentStackTrace` — the call `defaultPanic` makes a
// moment later, into a writer of its own — HANGS THE PROCESS when it is
// made from a panic handler BEFORE `defaultPanic` has run. Observed at 0%
// CPU in `futex_do_wait` with libc's debug info half-open, identically in a
// test binary and in a standalone build carrying this program's
// `std_options_debug_io`. Calling `debug_io.vtable.crashHandler` first does
// not help, and neither does holding `lockStderr` around it.
//
// `captureCurrentStackTrace` looks like the safe half of that and is NOT.
// `StackIterator.init` picks the `.di` strategy whenever `SelfInfo` can
// unwind, `stratOk` accepts `.di` no matter what `allow_unsafe_unwind`
// says, and `.di` unwinding takes `SelfInfo`'s rwlock EXCLUSIVELY on its
// first call — the only kind of call a panic record ever makes — while it
// runs `dl_iterate_phdr` and a DWARF CFI machine and allocates. A panic in
// there (a smashed stack is a leading cause of getting here at all) leaves
// that lock held forever, because the unlock is a `defer` in a frame that
// never returns. `defaultPanic` then asks for the same lock and waits on it
// for the life of the process: A CRASH BECOMES A HANG, which is strictly
// worse than the behaviour this file was added to improve on. What makes
// `defaultPanic` itself survive that is its own PRIVATE `panic_stage`,
// reachable from nowhere out here.
//
// So the frames stay on stderr, where `defaultPanic` prints them under the
// staging that makes them safe, and this file keeps what it can gather
// without asking the process any questions: which build, when, where, and
// what it said. That is the half a reporter cannot reconstruct afterwards.
w.print("panic: {s}\n", .{msg}) catch {};
// One `write(2)`, and the loop is for a short write rather than for a
// second record: `O_APPEND` puts it at the end whatever else is writing,
// and nothing above here can fail in a way that leaves it half-built.
const written = w.buffered();
var off: usize = 0;
while (off < written.len) {
const n = std.c.write(fd, written.ptr + off, written.len - off);
if (n <= 0) break; // including EINTR: the record is best effort
off += @intCast(n);
}
}
test "a crash record names the build, appends, and carries the panic message" {
const io = std.testing.io;
const gpa = std.testing.allocator;
var tmp = std.testing.tmpDir(.{});
defer tmp.cleanup();
var base_buf: [std.fs.max_path_bytes]u8 = undefined;
const base_len = try tmp.dir.realPath(io, &base_buf);
// A directory that does NOT exist yet, which is the case that matters: the
// user who has never written an init file is the likeliest reporter.
const config_dir = try std.fs.path.join(gpa, &.{ base_buf[0..base_len], "pardes" });
defer gpa.free(config_dir);
const saved_len = dir_len;
defer dir_len = saved_len;
setDir(config_dir);
record("first boom");
// Deliberately released, which the panic path never does — see `recording`.
// Two records is how the APPEND is proved, and this is the only caller
// allowed to ask for a second one.
recording.store(false, .seq_cst);
record("second boom");
const path = try std.fs.path.join(gpa, &.{ config_dir, name });
defer gpa.free(path);
const bytes = try std.Io.Dir.cwd().readFileAlloc(io, path, gpa, .limited(1 << 20));
defer gpa.free(bytes);
// The metadata line, then the message, for each of the two panics: the
// file is appended to, never rewritten.
try std.testing.expect(std.mem.count(u8, bytes, build_id ++ " ") == 2);
try std.testing.expect(std.mem.indexOf(u8, bytes, "panic: first boom\n") != null);
const second = std.mem.indexOf(u8, bytes, "panic: second boom\n").?;
try std.testing.expect(std.mem.indexOf(u8, bytes, "panic: first boom\n").? < second);
// ...and the platform and a UTC stamp on that line, which is the half a
// version string cannot give: `1970-` would mean the clock read failed.
const stamp = std.mem.indexOf(u8, bytes, @tagName(builtin.os.tag) ++ "-" ++ @tagName(builtin.cpu.arch)).?;
try std.testing.expect(stamp < second);
try std.testing.expect(std.mem.indexOf(u8, bytes, "1970-") == null);
// Two lines per record and no third: the frames deliberately do NOT come
// here (see `record`), and a future edit that starts collecting them again
// is the thing this pins. Counted rather than pattern-matched, because the
// failure being guarded is "something extra appeared", which no assertion
// about known content can see.
// Three newlines a record: the blank one that separates records, the end of
// the metadata line, and the end of the `panic:` line.
try std.testing.expectEqual(@as(usize, 6), std.mem.count(u8, bytes, "\n"));
// No directory, no file: a core that never resolved a config directory
// panics exactly as it did before this existed.
dir_len = 0;
recording.store(false, .seq_cst);
record("unrecorded");
const again = try std.Io.Dir.cwd().readFileAlloc(io, path, gpa, .limited(1 << 20));
defer gpa.free(again);
try std.testing.expectEqual(bytes.len, again.len);
}
|