summaryrefslogtreecommitdiff
path: root/src/crash.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/crash.zig')
-rw-r--r--src/crash.zig248
1 files changed, 248 insertions, 0 deletions
diff --git a/src/crash.zig b/src/crash.zig
new file mode 100644
index 00000000..10d38591
--- /dev/null
+++ b/src/crash.zig
@@ -0,0 +1,248 @@
+//! The crash file: what a panic leaves behind for the human who has to report it.
+//!
+//! stderr is where a panic has always gone, and stderr is the worst place this
+//! program has. In the tty shell stderr IS the screen — the trace lands on the
+//! grid the terminal is being reset out of, and the next redraw or the next
+//! `clear` takes it with it. The GUI and the AppKit shells have no terminal at
+//! all, so it goes to a console nobody has open. A detached session's goes
+//! wherever the launcher left it, which is usually /dev/null. So a record is
+//! appended to `<config dir>/crashes` as well, under one line naming the build
+//! that produced it: a bug report needs the version and the commit, and a user
+//! who runs pardes already knows where its config lives.
+//!
+//! The record is the metadata line and the panic message, and NOT the frames:
+//! every way of collecting those from out here can wedge the process instead of
+//! ending it. `record` says exactly why, at length, because the reasoning is
+//! not visible from the call it does not make.
+//!
+//! Panics only. A SIGSEGV never reaches a panic handler (main.zig `debug`
+//! hooks that path separately) and unwinding one needs the signal's saved cpu
+//! context, which is a different job than this one.
+
+const std = @import("std");
+const builtin = @import("builtin");
+const pardes = @import("pardes.zig");
+
+/// Beside `init`, so `Config` opens the directory that holds both.
+pub const name = "crashes";
+
+/// The config directory `config.User.load` resolved, COPIED rather than
+/// borrowed: main.zig hands over an arena slice that outlives the process, but
+/// the AppKit host's lives in a `config_arena` its own `errdefer` frees on a
+/// failed init and its teardown frees at quit — and a panic after either would
+/// have built a path out of freed memory and then created a directory at it.
+/// A panic handler is the one caller that cannot check whether its input is
+/// still alive, so it does not borrow.
+var dir_buf: [1024]u8 = undefined;
+var dir_len: usize = 0;
+
+/// Called where the launcher sets `Options.config_dir`. Ignores a path too long
+/// to hold, which is a crash file that never appears rather than a truncated
+/// path pointing somewhere real.
+pub fn setDir(path: []const u8) void {
+ if (path.len == 0 or path.len > dir_buf.len) return;
+ @memcpy(dir_buf[0..path.len], path);
+ dir_len = path.len;
+}
+
+/// Built at comptime out of the two halves main.zig's `--version` prints, and
+/// for the same reason: there is nothing here to format at runtime. A build
+/// from a tarball says `pardes 0.0.1` rather than inventing a revision.
+const build_id = if (pardes.commit) |c|
+ "pardes " ++ pardes.version ++ " (" ++ c ++ ")"
+else
+ "pardes " ++ pardes.version;
+
+/// The whole record is formatted here before a byte of it is written, because
+/// the fd is opened `O_APPEND` and one `write(2)` of the lot is what keeps two
+/// crashing processes from interleaving their records line by line. Static
+/// rather than a local: a panic handler may be running on a thread stack of a
+/// few pages. Two lines fit several times over — the panic message is the only
+/// part with no bound of its own, and a longer one is simply cut.
+var scratch: [8 << 10]u8 = undefined;
+
+/// One record per process, taken by whoever panics first. `defaultPanic` gets
+/// the same property from a private `panic_stage` and from the stderr lock, and
+/// neither is reachable from out here: two threads panicking at once would
+/// interleave `@memcpy`s into one `scratch` and each write out the mixture, and
+/// the nested panic `defaultPanic`'s own trace printer can raise comes back
+/// through the handler and appends its wreckage under the real message. Both
+/// end here instead, and the loser falls through to `defaultPanic`, which still
+/// says everything on stderr.
+var recording: std.atomic.Value(bool) = .init(false);
+
+/// Append one crash record, or silently do nothing. Called FROM the panic
+/// handler, so it takes no lock of this program's, allocates nothing of its
+/// own, and does its own file I/O: a static buffer, one `open`, one `write`.
+/// Every failure is swallowed — a crash file that could not be written must not
+/// become the crash, and stderr is about to get the message either way.
+pub fn record(msg: []const u8) void {
+ if (dir_len == 0) return;
+ // ONCE PER PROCESS, and never released. A process panics once and dies; a
+ // SECOND record is always the wreckage of the first. Observed with the
+ // flag released on the way out: `defaultPanic`'s own trace printing raised
+ // a nested panic, which came back through the handler and appended
+ // "panic: reached unreachable code" under the real message — burying the
+ // one line the file exists to keep. A test resets it deliberately.
+ if (recording.swap(true, .seq_cst)) return;
+ const config_dir = dir_buf[0..dir_len];
+ var path_buf: [1024:0]u8 = undefined;
+ if (config_dir.len + 1 + name.len >= path_buf.len) return;
+
+ // The directory is where the config WOULD be, not where it is: a user who
+ // never wrote an init file has no config directory, and the first thing
+ // they have to report is this. Parents are assumed and EEXIST is the
+ // normal answer, exactly as dump.zig bets for its own directory.
+ @memcpy(path_buf[0..config_dir.len], config_dir);
+ path_buf[config_dir.len] = 0;
+ _ = std.c.mkdir(path_buf[0..config_dir.len :0], 0o755);
+
+ path_buf[config_dir.len] = '/';
+ @memcpy(path_buf[config_dir.len + 1 ..][0..name.len], name);
+ path_buf[config_dir.len + 1 + name.len] = 0;
+ const path = path_buf[0 .. config_dir.len + 1 + name.len :0];
+
+ const fd = std.c.open(path, .{
+ .ACCMODE = .WRONLY,
+ .CREAT = true,
+ .APPEND = true,
+ }, @as(std.c.mode_t, 0o600));
+ if (fd < 0) return;
+ defer _ = std.c.close(fd);
+
+ var w: std.Io.Writer = .fixed(&scratch);
+
+ // The metadata line. Same clock and the same civil-time arithmetic as
+ // dump.zig's filename, printed as UTC because a crash file is read by
+ // whoever the reporter sends it to and their zone is not the reporter's.
+ //
+ // The return IS checked, unlike dump.zig's, because this one runs where a
+ // panic cannot be afforded: a failed `clock_gettime` leaves `ts` undefined,
+ // and an undefined large-positive `sec` walks `calculateYearDay`'s `u16`
+ // year past 65535 and overflow-panics INSIDE the panic handler. Debug fills
+ // it with 0xaa and lands in 1970, which is why it reads as harmless.
+ var ts: std.c.timespec = .{ .sec = 0, .nsec = 0 };
+ if (std.c.clock_gettime(.REALTIME, &ts) != 0) ts = .{ .sec = 0, .nsec = 0 };
+ const es: std.time.epoch.EpochSeconds = .{ .secs = @intCast(@max(0, ts.sec)) };
+ const yd = es.getEpochDay().calculateYearDay();
+ const md = yd.calculateMonthDay();
+ const ds = es.getDaySeconds();
+ w.print("\n" ++ build_id ++ " {d:0>4}-{d:0>2}-{d:0>2}T{d:0>2}:{d:0>2}:{d:0>2}Z {s}-{s} pid {d}\n", .{
+ yd.year,
+ md.month.numeric(),
+ @as(u8, md.day_index) + 1,
+ ds.getHoursIntoDay(),
+ ds.getMinutesIntoHour(),
+ ds.getSecondsIntoMinute(),
+ @tagName(builtin.os.tag),
+ @tagName(builtin.cpu.arch),
+ // Unsigned: `{d}` prints a leading '+' for a positive SIGNED int, which
+ // is the same note main.zig makes where it names a `--detach` session
+ // after this pid.
+ @as(u32, @intCast(std.c.getpid())),
+ }) catch {};
+
+ // ...and under it the line stderr is about to print.
+ //
+ // NO STACK TRACE, AND NOT EVEN THE RAW RETURN ADDRESSES. This is the whole
+ // design decision in this file, and it was arrived at by measurement, so it
+ // is written down here rather than left to be rediscovered.
+ //
+ // `std.debug.writeCurrentStackTrace` — the call `defaultPanic` makes a
+ // moment later, into a writer of its own — HANGS THE PROCESS when it is
+ // made from a panic handler BEFORE `defaultPanic` has run. Observed at 0%
+ // CPU in `futex_do_wait` with libc's debug info half-open, identically in a
+ // test binary and in a standalone build carrying this program's
+ // `std_options_debug_io`. Calling `debug_io.vtable.crashHandler` first does
+ // not help, and neither does holding `lockStderr` around it.
+ //
+ // `captureCurrentStackTrace` looks like the safe half of that and is NOT.
+ // `StackIterator.init` picks the `.di` strategy whenever `SelfInfo` can
+ // unwind, `stratOk` accepts `.di` no matter what `allow_unsafe_unwind`
+ // says, and `.di` unwinding takes `SelfInfo`'s rwlock EXCLUSIVELY on its
+ // first call — the only kind of call a panic record ever makes — while it
+ // runs `dl_iterate_phdr` and a DWARF CFI machine and allocates. A panic in
+ // there (a smashed stack is a leading cause of getting here at all) leaves
+ // that lock held forever, because the unlock is a `defer` in a frame that
+ // never returns. `defaultPanic` then asks for the same lock and waits on it
+ // for the life of the process: A CRASH BECOMES A HANG, which is strictly
+ // worse than the behaviour this file was added to improve on. What makes
+ // `defaultPanic` itself survive that is its own PRIVATE `panic_stage`,
+ // reachable from nowhere out here.
+ //
+ // So the frames stay on stderr, where `defaultPanic` prints them under the
+ // staging that makes them safe, and this file keeps what it can gather
+ // without asking the process any questions: which build, when, where, and
+ // what it said. That is the half a reporter cannot reconstruct afterwards.
+ w.print("panic: {s}\n", .{msg}) catch {};
+
+ // One `write(2)`, and the loop is for a short write rather than for a
+ // second record: `O_APPEND` puts it at the end whatever else is writing,
+ // and nothing above here can fail in a way that leaves it half-built.
+ const written = w.buffered();
+ var off: usize = 0;
+ while (off < written.len) {
+ const n = std.c.write(fd, written.ptr + off, written.len - off);
+ if (n <= 0) break; // including EINTR: the record is best effort
+ off += @intCast(n);
+ }
+}
+
+test "a crash record names the build, appends, and carries the panic message" {
+ const io = std.testing.io;
+ const gpa = std.testing.allocator;
+ var tmp = std.testing.tmpDir(.{});
+ defer tmp.cleanup();
+ var base_buf: [std.fs.max_path_bytes]u8 = undefined;
+ const base_len = try tmp.dir.realPath(io, &base_buf);
+
+ // A directory that does NOT exist yet, which is the case that matters: the
+ // user who has never written an init file is the likeliest reporter.
+ const config_dir = try std.fs.path.join(gpa, &.{ base_buf[0..base_len], "pardes" });
+ defer gpa.free(config_dir);
+ const saved_len = dir_len;
+ defer dir_len = saved_len;
+ setDir(config_dir);
+
+ record("first boom");
+ // Deliberately released, which the panic path never does — see `recording`.
+ // Two records is how the APPEND is proved, and this is the only caller
+ // allowed to ask for a second one.
+ recording.store(false, .seq_cst);
+ record("second boom");
+
+ const path = try std.fs.path.join(gpa, &.{ config_dir, name });
+ defer gpa.free(path);
+ const bytes = try std.Io.Dir.cwd().readFileAlloc(io, path, gpa, .limited(1 << 20));
+ defer gpa.free(bytes);
+
+ // The metadata line, then the message, for each of the two panics: the
+ // file is appended to, never rewritten.
+ try std.testing.expect(std.mem.count(u8, bytes, build_id ++ " ") == 2);
+ try std.testing.expect(std.mem.indexOf(u8, bytes, "panic: first boom\n") != null);
+ const second = std.mem.indexOf(u8, bytes, "panic: second boom\n").?;
+ try std.testing.expect(std.mem.indexOf(u8, bytes, "panic: first boom\n").? < second);
+ // ...and the platform and a UTC stamp on that line, which is the half a
+ // version string cannot give: `1970-` would mean the clock read failed.
+ const stamp = std.mem.indexOf(u8, bytes, @tagName(builtin.os.tag) ++ "-" ++ @tagName(builtin.cpu.arch)).?;
+ try std.testing.expect(stamp < second);
+ try std.testing.expect(std.mem.indexOf(u8, bytes, "1970-") == null);
+
+ // Two lines per record and no third: the frames deliberately do NOT come
+ // here (see `record`), and a future edit that starts collecting them again
+ // is the thing this pins. Counted rather than pattern-matched, because the
+ // failure being guarded is "something extra appeared", which no assertion
+ // about known content can see.
+ // Three newlines a record: the blank one that separates records, the end of
+ // the metadata line, and the end of the `panic:` line.
+ try std.testing.expectEqual(@as(usize, 6), std.mem.count(u8, bytes, "\n"));
+
+ // No directory, no file: a core that never resolved a config directory
+ // panics exactly as it did before this existed.
+ dir_len = 0;
+ recording.store(false, .seq_cst);
+ record("unrecorded");
+ const again = try std.Io.Dir.cwd().readFileAlloc(io, path, gpa, .limited(1 << 20));
+ defer gpa.free(again);
+ try std.testing.expectEqual(bytes.len, again.len);
+}