summaryrefslogtreecommitdiff
path: root/introspect/src/linux/debug.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-09-19 21:26:05 -0300
committerGabriel Schneider <[email protected]>2026-09-19 21:26:05 -0300
commitb05abcba3ea09ea106ad28364c6e40a3ec31b890 (patch)
tree9170fac5e7e5d8bde108de34a182aaa9d6844117 /introspect/src/linux/debug.zig
parentae310a207534b33b7321dd2b9f423a73b1969159 (diff)
downloadcloud9-b05abcba3ea09ea106ad28364c6e40a3ec31b890.tar.gz
cloud9-b05abcba3ea09ea106ad28364c6e40a3ec31b890.zip
Add 9player and introspect as programs beside the library
9player/: FUSE mount CLI that mounts a 9P2000 tree into a fresh user+mount namespace and runs a program in it (no root, no libfuse, no libc). introspect/: the 9P debug/introspection library (freestanding core, value renderers, Linux probe with threads/stacks/memory/breakpoints/panics) and its demo server. Each has its own build fragment; the root build.zig wires them behind -D9player/-Dintrospect with namespaced steps (9player-itest, introspect-check-freestanding, programs-test, ...) and exports the introspect module for dependents. This is the layout for related programs. Co-Authored-By: Claude Fable 5.1 <[email protected]>
Diffstat (limited to 'introspect/src/linux/debug.zig')
-rw-r--r--introspect/src/linux/debug.zig1458
1 files changed, 1458 insertions, 0 deletions
diff --git a/introspect/src/linux/debug.zig b/introspect/src/linux/debug.zig
new file mode 100644
index 0000000..4d43d5b
--- /dev/null
+++ b/introspect/src/linux/debug.zig
@@ -0,0 +1,1458 @@
+//! Linux debug facilities for the introspect server: threads, stacks,
+//! registers, address → source, memory, breakpoints and panics.
+//!
+//! This file is a pure API; a later adapter turns it into a core `Provider`.
+//! All text is written to a `*std.Io.Writer`. Nothing here allocates after
+//! `init` except from the caller-provided `text_buf`, which is used as a fixed
+//! arena for symbol text and reset before every query.
+//!
+//! Only one `Debug` may exist per process: the signal handlers and the panic
+//! hook find their state through the global `current` pointer set by `init`.
+//!
+//! Mechanics
+//!
+//! * Capturing another thread's stack or registers: the calling (server)
+//! thread sends `capture_signal` with `tgkill`. The SA_SIGINFO handler copies
+//! the interrupted register state (`cpu_context.fromPosixSignalContext`) into
+//! the single capture slot and parks on a futex. The server unwinds the
+//! parked thread's stack from that context, releases the target, then
+//! symbolizes. The handler is async-signal-safe: no allocation, no
+//! `std.debug`, no locks other than the futex. A target that does not run
+//! the handler within `capture_timeout_ns` (signal masked, thread in D
+//! state, ...) yields `error.Timeout`; a late-arriving handler run cannot
+//! corrupt a reused slot because it must match the requested tid and win a
+//! compare-and-swap from `armed` on the slot state (that pair plays the role
+//! of a generation counter: a stale run finds the slot idle, armed for
+//! another tid, or armed for itself, in which case its capture is simply the
+//! valid answer to the new request).
+//! * Breakpoints: `@breakpoint()` raises SIGTRAP on the executing thread only.
+//! The handler claims a pause slot, saves the context and parks on a futex
+//! until `resumeThread`. On x86_64 the saved PC is already past `int3`; on
+//! aarch64 the handler advances PC by 4 in the ucontext before returning
+//! (only for a real `brk`, i.e. a kernel-generated si_code; a SIGTRAP sent
+//! with kill/tgkill parks the thread where it was). With no free slot the
+//! thread steps over the breakpoint and keeps running (`traps_skipped`
+//! counts them): the debug layer never kills the process. The server thread
+//! itself (`server_tid`) is never parked, a breakpoint there is stepped
+//! over, because nobody could resume it. Only a stale handler run after
+//! `deinit` (no `current`) falls back to the default disposition.
+//! * Panics: `panicHook` records the message and a stack capture, then, if
+//! `hold_on_panic` and a `Debug` exists, parks until `panicContinue`; then
+//! `std.debug.defaultPanic` runs. A nested or second panic, or a panic on
+//! the server thread itself (which could never be continued), goes
+//! straight to the default handler.
+//! * std.debug's `SelfInfo` guards its state with an `Io.RwLock`. A target
+//! parked while holding it (a thread inside a stack-trace dump, say) would
+//! deadlock the unwind, so after parking a thread the lock is probed with
+//! `tryLock`; a held lock yields `error.Busy` and the target is released.
+//! * Known-module guard: `std.debug.SelfInfo` (Zig 0.16) rebuilds its module
+//! list whenever it is asked about an address outside every known module,
+//! freeing the CIE lists its unwind cache still points into; later unwinds
+//! then read freed memory. `init` records the PT_LOAD ranges of the
+//! executable (the same source std uses) and every lookup or unwind is
+//! first checked against them; addresses outside (unmapped, vDSO, ...)
+//! render as "?" and are never handed to std.
+
+const std = @import("std");
+const builtin = @import("builtin");
+const linux = std.os.linux;
+const cpu_context = std.debug.cpu_context;
+const Writer = std.Io.Writer;
+const Native = cpu_context.Native;
+const arch = builtin.cpu.arch;
+
+pub const Options = struct {
+ /// Used for `std.debug` symbolization (reading debug info from disk).
+ io: std.Io,
+ /// Fixed arena for symbol text. A `FixedBufferAllocator` is placed over it
+ /// and reset before every query. 16 KiB is plenty; 4 KiB is a sane floor.
+ text_buf: []u8,
+ /// Real-time signal used to snapshot other threads. SIGRTMIN is 32 on
+ /// Linux without libc; the default is SIGRTMIN+3.
+ capture_signal: u8 = default_capture_signal,
+ /// How long to wait for a target thread to run the capture handler.
+ capture_timeout_ns: u64 = 250 * std.time.ns_per_ms,
+ /// How many threads may be parked in `@breakpoint()` at once (≤ 32).
+ max_paused: u8 = 16,
+};
+
+pub const default_capture_signal: u8 = 32 + 3;
+
+/// Hard upper bound of `Options.max_paused` (slot storage is static).
+pub const max_paused_cap = 32;
+/// Maximum number of frames written by any stack function.
+pub const max_frames = 64;
+/// Maximum number of tids enumerated from /proc/self/task.
+pub const max_threads = 512;
+/// Upper bound of the recorded panic message.
+pub const panic_msg_cap = 1024;
+/// Maximum number of PT_LOAD ranges recorded by the known-module guard.
+pub const max_ranges = 64;
+
+/// Consulted by `panicHook`: when true and a `Debug` is initialized, the
+/// panicking thread is held until `panicContinue`.
+pub var hold_on_panic: bool = true;
+
+/// The one live instance, set by `init`, cleared by `deinit`.
+pub var current: ?*Debug = null;
+
+/// The tid of the thread serving requests (0 = none). That thread is never
+/// parked by a breakpoint or held by a panic, since nobody could release it.
+pub var server_tid: std.atomic.Value(u32) = .init(0);
+
+/// Breakpoints stepped over because no pause slot was free, or because they
+/// were hit on the server thread.
+pub var traps_skipped: std.atomic.Value(u32) = .init(0);
+
+pub const Error = error{
+ /// The target thread did not run the capture handler in time.
+ Timeout,
+ /// No thread with that tid exists in this process.
+ NoThread,
+ /// The address is not mapped (EFAULT from process_vm_readv/writev).
+ Unmapped,
+ /// The thread is not parked in a breakpoint.
+ NotPaused,
+ /// No panic has been recorded / is being held.
+ NoPanic,
+ /// Another `Debug` already exists in this process.
+ AlreadyInitialized,
+ /// The operation is not available on this architecture / kernel.
+ Unsupported,
+ /// The target thread is parked inside std.debug (holding its lock); its
+ /// stack cannot be unwound without deadlocking. Retry later.
+ Busy,
+ /// Invalid option value.
+ InvalidOptions,
+ /// A syscall or /proc read failed unexpectedly.
+ Unexpected,
+ /// The writer failed.
+ WriteFailed,
+};
+
+// Capture slot states.
+const cap_idle: u32 = 0;
+const cap_armed: u32 = 1;
+const cap_capturing: u32 = 2;
+const cap_captured: u32 = 3;
+const cap_failed: u32 = 4;
+
+// Pause slot states.
+const pause_free: u32 = 0;
+const pause_claimed: u32 = 1;
+const pause_paused: u32 = 2;
+const pause_resuming: u32 = 3;
+
+const CaptureSlot = struct {
+ state: std.atomic.Value(u32) = .init(cap_idle),
+ target_tid: std.atomic.Value(u32) = .init(0),
+ ctx: Native = undefined,
+};
+
+const PauseSlot = struct {
+ state: std.atomic.Value(u32) = .init(pause_free),
+ tid: std.atomic.Value(u32) = .init(0),
+ ctx: Native = undefined,
+};
+
+pub const Debug = struct {
+ io: std.Io,
+ text_buf: []u8,
+ capture_signal: linux.SIG,
+ capture_timeout_ns: u64,
+ max_paused: u8,
+
+ capture: CaptureSlot = .{},
+ paused: [max_paused_cap]PauseSlot = [_]PauseSlot{.{}} ** max_paused_cap,
+
+ old_capture_action: linux.Sigaction = undefined,
+ old_trap_action: linux.Sigaction = undefined,
+ breakpoints_enabled: bool = false,
+
+ tids: [max_threads]u32 = undefined,
+ tid_count: usize = 0,
+
+ ranges: [max_ranges]Range = undefined,
+ range_count: usize = 0,
+
+ const Range = struct { start: usize, len: usize };
+
+ /// Installs the capture handler (not the SIGTRAP handler) and publishes
+ /// `d` as `current`.
+ pub fn init(d: *Debug, opts: Options) Error!void {
+ if (current != null) return error.AlreadyInitialized;
+ if (opts.capture_signal < 32 or opts.capture_signal >= linux.NSIG) return error.InvalidOptions;
+ if (opts.max_paused == 0 or opts.max_paused > max_paused_cap) return error.InvalidOptions;
+ if (Native == noreturn) return error.Unsupported;
+ d.* = .{
+ .io = opts.io,
+ .text_buf = opts.text_buf,
+ .capture_signal = @enumFromInt(opts.capture_signal),
+ .capture_timeout_ns = opts.capture_timeout_ns,
+ .max_paused = opts.max_paused,
+ };
+ d.scanModules();
+ const act: linux.Sigaction = .{
+ .handler = .{ .sigaction = captureHandler },
+ .mask = linux.sigemptyset(),
+ .flags = linux.SA.SIGINFO | linux.SA.RESTART,
+ };
+ current = d;
+ if (linux.errno(linux.sigaction(d.capture_signal, &act, &d.old_capture_action)) != .SUCCESS) {
+ current = null;
+ return error.Unexpected;
+ }
+ }
+
+ /// Restores the signal dispositions and clears `current`. Threads parked
+ /// in a breakpoint are resumed first.
+ pub fn deinit(d: *Debug) void {
+ d.disableBreakpoints();
+ _ = linux.sigaction(d.capture_signal, &d.old_capture_action, null);
+ if (current == d) current = null;
+ }
+
+ /// Installs the SIGTRAP handler so that `@breakpoint()` parks the thread.
+ pub fn enableBreakpoints(d: *Debug) Error!void {
+ if (d.breakpoints_enabled) return;
+ if (arch != .x86_64 and !arch.isAARCH64()) return error.Unsupported;
+ const act: linux.Sigaction = .{
+ .handler = .{ .sigaction = trapHandler },
+ .mask = linux.sigemptyset(),
+ .flags = linux.SA.SIGINFO | linux.SA.RESTART,
+ };
+ if (linux.errno(linux.sigaction(.TRAP, &act, &d.old_trap_action)) != .SUCCESS) return error.Unexpected;
+ d.breakpoints_enabled = true;
+ }
+
+ /// Restores the previous SIGTRAP disposition and resumes every parked thread.
+ pub fn disableBreakpoints(d: *Debug) void {
+ if (!d.breakpoints_enabled) return;
+ _ = linux.sigaction(.TRAP, &d.old_trap_action, null);
+ d.breakpoints_enabled = false;
+ for (&d.paused) |*slot| {
+ if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) == null)
+ futexWake(&slot.state);
+ }
+ }
+
+ // ---------------------------------------------------------------- threads
+
+ /// The nth tid of this process, numerically sorted; null past the end.
+ /// Index 0 rescans /proc/self/task; higher indices reuse that scan.
+ pub fn threadAt(d: *Debug, index: usize) ?u32 {
+ if (index == 0 or d.tid_count == 0) d.scanThreads();
+ if (index >= d.tid_count) return null;
+ return d.tids[index];
+ }
+
+ pub fn threadExists(d: *Debug, tid: u32) bool {
+ _ = d;
+ var path_buf: [64]u8 = undefined;
+ const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return false;
+ var buf: [32]u8 = undefined;
+ _ = readFile(path, &buf) catch return false;
+ return true;
+ }
+
+ /// The thread's comm (without the trailing newline).
+ pub fn threadName(d: *Debug, tid: u32, w: *Writer) Error!void {
+ _ = d;
+ var path_buf: [64]u8 = undefined;
+ const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return error.Unexpected;
+ var buf: [64]u8 = undefined;
+ const text = readFile(path, &buf) catch |err| switch (err) {
+ error.NotFound => return error.NoThread,
+ else => return error.Unexpected,
+ };
+ w.writeAll(std.mem.trimEnd(u8, text, "\n")) catch return error.WriteFailed;
+ }
+
+ /// A few fields of /proc/self/task/<tid>/stat, one "name value" per line:
+ /// state, utime, stime, minflt, majflt, priority, nice, processor.
+ pub fn threadStat(d: *Debug, tid: u32, w: *Writer) Error!void {
+ _ = d;
+ var path_buf: [64]u8 = undefined;
+ const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/stat", .{tid}) catch return error.Unexpected;
+ var buf: [1024]u8 = undefined;
+ const text = readFile(path, &buf) catch |err| switch (err) {
+ error.NotFound => return error.NoThread,
+ else => return error.Unexpected,
+ };
+ // "<pid> (<comm>) S <fields...>"; comm may contain spaces and parens.
+ const close = std.mem.lastIndexOfScalar(u8, text, ')') orelse return error.Unexpected;
+ var it = std.mem.tokenizeScalar(u8, text[close + 1 ..], ' ');
+ // Field numbers below are 0-based from `state`.
+ const wanted = [_]struct { idx: usize, name: []const u8 }{
+ .{ .idx = 0, .name = "state" },
+ .{ .idx = 11, .name = "utime" },
+ .{ .idx = 12, .name = "stime" },
+ .{ .idx = 7, .name = "minflt" },
+ .{ .idx = 9, .name = "majflt" },
+ .{ .idx = 15, .name = "priority" },
+ .{ .idx = 16, .name = "nice" },
+ .{ .idx = 36, .name = "processor" },
+ };
+ var fields: [40][]const u8 = undefined;
+ var n: usize = 0;
+ while (it.next()) |f| : (n += 1) {
+ if (n == fields.len) break;
+ fields[n] = f;
+ }
+ for (wanted) |want| {
+ const value = if (want.idx < n) fields[want.idx] else "?";
+ w.print("{s} {s}\n", .{ want.name, value }) catch return error.WriteFailed;
+ }
+ }
+
+ /// "#n 0x<addr> in <fn> (<file>:<line>:<col>)" per frame. The calling
+ /// thread unwinds itself directly; any other thread is captured with the
+ /// capture signal.
+ pub fn threadStack(d: *Debug, tid: u32, w: *Writer) Error!void {
+ var addrs: [max_frames]usize = undefined;
+ var trace: std.debug.StackTrace = undefined;
+ if (tid == selfTid()) {
+ trace = std.debug.captureCurrentStackTrace(.{}, &addrs);
+ } else {
+ try d.captureThread(tid);
+ if (!d.selfInfoFree()) {
+ d.releaseCapture();
+ return error.Busy;
+ }
+ trace = d.unwindContext(&d.capture.ctx, &addrs);
+ d.releaseCapture();
+ }
+ try d.writeFrames(trace.return_addresses, w);
+ }
+
+ /// "<reg> 0x<hex>" per general register, plus pc/sp/fp aliases.
+ pub fn threadRegs(d: *Debug, tid: u32, w: *Writer) Error!void {
+ if (tid == selfTid()) {
+ const ctx = Native.current();
+ return writeRegs(&ctx, w);
+ }
+ try d.captureThread(tid);
+ const ctx = d.capture.ctx;
+ d.releaseCapture();
+ return writeRegs(&ctx, w);
+ }
+
+ // ------------------------------------------------------ addresses & memory
+
+ /// "fn\nfile:line:col\nmodule\n", unknown parts as "?".
+ pub fn resolveAddr(d: *Debug, addr: usize, w: *Writer) Error!void {
+ if (!d.knownCode(addr)) return w.writeAll("?\n?\n?\n") catch error.WriteFailed;
+ var fba = std.heap.FixedBufferAllocator.init(d.text_buf);
+ const alloc = fba.allocator();
+ const di = std.debug.getSelfDebugInfo() catch return error.Unsupported;
+ var sym = std.debug.Symbol.unknown;
+ var symbols: std.ArrayList(std.debug.Symbol) = .empty;
+ if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) {
+ if (symbols.items.len > 0) sym = symbols.items[0];
+ } else |_| {}
+ w.print("{s}\n", .{sym.name orelse "?"}) catch return error.WriteFailed;
+ if (sym.source_location) |sl| {
+ w.print("{s}:{d}:{d}\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed;
+ } else {
+ w.writeAll("?\n") catch return error.WriteFailed;
+ }
+ const module = di.getModuleName(d.io, addr) catch "?";
+ w.print("{s}\n", .{module}) catch return error.WriteFailed;
+ }
+
+ /// Reads `buf.len` bytes at `addr` via process_vm_readv on the own
+ /// process. Never faults. Returns the number of bytes read (short when the
+ /// range crosses into an unmapped page); `error.Unmapped` when nothing
+ /// could be read.
+ pub fn readMem(d: *Debug, addr: usize, buf: []u8) Error!usize {
+ _ = d;
+ if (buf.len == 0) return 0;
+ // Page 0 is never mapped (mmap_min_addr) and a null `iovec.base` is a
+ // safety-checked cast; the same answer without the trap.
+ if (addr == 0) return error.Unmapped;
+ const local = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }};
+ const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = buf.len }};
+ const rc = linux.process_vm_readv(linux.getpid(), &local, &remote, 0);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return rc,
+ .FAULT => return error.Unmapped,
+ .NOSYS, .PERM => return error.Unsupported,
+ else => return error.Unexpected,
+ }
+ }
+
+ /// Writes `data` at `addr` via process_vm_writev. Read-only mappings also
+ /// report `error.Unmapped` (the kernel says EFAULT for both).
+ pub fn writeMem(d: *Debug, addr: usize, data: []const u8) Error!usize {
+ _ = d;
+ if (data.len == 0) return 0;
+ if (addr == 0) return error.Unmapped;
+ const local = [_]std.posix.iovec_const{.{ .base = data.ptr, .len = data.len }};
+ const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = data.len }};
+ const rc = linux.process_vm_writev(linux.getpid(), &local, &remote, 0);
+ switch (linux.errno(rc)) {
+ .SUCCESS => return rc,
+ .FAULT => return error.Unmapped,
+ .NOSYS, .PERM => return error.Unsupported,
+ else => return error.Unexpected,
+ }
+ }
+
+ /// Hexdump of `len` bytes at `addr` in the shape of `std.debug.dumpHex`
+ /// (16 bytes per line, address column, bytes in two groups, ASCII column).
+ /// Stops early at the first unmapped byte; `error.Unmapped` only when the
+ /// very first chunk is unreadable.
+ pub fn hexdump(d: *Debug, addr: usize, len: usize, w: *Writer) Error!void {
+ var chunk: [256]u8 = undefined;
+ var done: usize = 0;
+ while (done < len) {
+ const want = @min(chunk.len, len - done);
+ const got = d.readMem(addr +% done, chunk[0..want]) catch |err| switch (err) {
+ error.Unmapped => if (done == 0) return error.Unmapped else break,
+ else => return err,
+ };
+ if (got == 0) break;
+ try writeHexLines(addr +% done, chunk[0..got], w);
+ done += got;
+ if (got < want) break;
+ }
+ }
+
+ /// Copies /proc/self/maps to `w`.
+ pub fn maps(d: *Debug, w: *Writer) Error!void {
+ _ = d;
+ return streamFile("/proc/self/maps", w);
+ }
+
+ /// Reads `buf.len` bytes of /proc/self/maps at `offset` (0 at the end).
+ /// Not a consistent snapshot across reads; a map appearing between two
+ /// reads shifts the text, like `cat` on /proc itself.
+ pub fn readMaps(d: *Debug, offset: u64, buf: []u8) Error!usize {
+ _ = d;
+ if (offset > std.math.maxInt(i64)) return 0;
+ return preadFile("/proc/self/maps", offset, buf);
+ }
+
+ // ------------------------------------------------------------ breakpoints
+
+ /// The nth tid currently parked in `@breakpoint()`.
+ pub fn pausedAt(d: *Debug, index: usize) ?u32 {
+ var n: usize = 0;
+ for (d.paused[0..d.max_paused]) |*slot| {
+ if (slot.state.load(.acquire) != pause_paused) continue;
+ if (n == index) return slot.tid.load(.acquire);
+ n += 1;
+ }
+ return null;
+ }
+
+ pub fn isPaused(d: *Debug, tid: u32) bool {
+ return d.pausedSlot(tid) != null;
+ }
+
+ pub fn pausedStack(d: *Debug, tid: u32, w: *Writer) Error!void {
+ const slot = d.pausedSlot(tid) orelse return error.NotPaused;
+ if (!d.selfInfoFree()) return error.Busy;
+ var addrs: [max_frames]usize = undefined;
+ const trace = d.unwindContext(&slot.ctx, &addrs);
+ try d.writeFrames(trace.return_addresses, w);
+ }
+
+ pub fn pausedRegs(d: *Debug, tid: u32, w: *Writer) Error!void {
+ const slot = d.pausedSlot(tid) orelse return error.NotPaused;
+ return writeRegs(&slot.ctx, w);
+ }
+
+ /// Lets a parked thread continue past its breakpoint.
+ pub fn resumeThread(d: *Debug, tid: u32) Error!void {
+ const slot = d.pausedSlot(tid) orelse return error.NotPaused;
+ if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) != null) return error.NotPaused;
+ futexWake(&slot.state);
+ }
+
+ fn pausedSlot(d: *Debug, tid: u32) ?*PauseSlot {
+ for (d.paused[0..d.max_paused]) |*slot| {
+ if (slot.state.load(.acquire) == pause_paused and slot.tid.load(.acquire) == tid) return slot;
+ }
+ return null;
+ }
+
+ // ------------------------------------------------------------------ panic
+
+ /// The recorded panic message; nothing before any panic.
+ pub fn panicMessage(d: *Debug, w: *Writer) Error!void {
+ _ = d;
+ if (panic_state.load(.acquire) == panic_none) return;
+ w.writeAll(panic_msg[0..panic_msg_len]) catch return error.WriteFailed;
+ }
+
+ /// Frames of the panicking thread, symbolized lazily.
+ pub fn panicStack(d: *Debug, w: *Writer) Error!void {
+ if (panic_state.load(.acquire) == panic_none) return;
+ try d.writeFrames(panic_addrs[0..panic_addr_count], w);
+ }
+
+ /// True while a panicking thread is parked waiting for `panicContinue`.
+ pub fn panicHeld(d: *Debug) bool {
+ _ = d;
+ return panic_state.load(.acquire) == panic_held;
+ }
+
+ /// Releases the held panicking thread into `std.debug.defaultPanic`.
+ pub fn panicContinue(d: *Debug) Error!void {
+ _ = d;
+ if (panic_state.cmpxchgStrong(panic_held, panic_continued, .acq_rel, .acquire) != null) return error.NoPanic;
+ futexWake(&panic_state);
+ }
+
+ // -------------------------------------------------------------- internals
+
+ fn scanThreads(d: *Debug) void {
+ d.tid_count = 0;
+ const fd_rc = linux.open("/proc/self/task", .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0);
+ if (linux.errno(fd_rc) != .SUCCESS) return;
+ const fd: i32 = @intCast(fd_rc);
+ defer _ = linux.close(fd);
+ var buf: [4096]u8 align(@alignOf(linux.dirent64)) = undefined;
+ while (true) {
+ const rc = linux.getdents64(fd, &buf, buf.len);
+ if (linux.errno(rc) != .SUCCESS or rc == 0) break;
+ var off: usize = 0;
+ while (off < rc) {
+ const ent: *align(1) const linux.dirent64 = @ptrCast(&buf[off]);
+ const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]);
+ const name = std.mem.span(name_ptr);
+ if (std.fmt.parseInt(u32, name, 10)) |tid| {
+ if (d.tid_count < max_threads) {
+ d.tids[d.tid_count] = tid;
+ d.tid_count += 1;
+ }
+ } else |_| {}
+ off += ent.reclen;
+ }
+ }
+ std.mem.sort(u32, d.tids[0..d.tid_count], {}, std.sort.asc(u32));
+ }
+
+ /// Arms the capture slot for `tid`, signals it and waits until the handler
+ /// has parked with its context copied. On success the caller owns the
+ /// slot until `releaseCapture`.
+ fn captureThread(d: *Debug, tid: u32) Error!void {
+ const slot = &d.capture;
+ slot.target_tid.store(tid, .release);
+ slot.state.store(cap_armed, .release);
+ const rc = linux.tgkill(linux.getpid(), @intCast(tid), d.capture_signal);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .SRCH => {
+ slot.state.store(cap_idle, .release);
+ return error.NoThread;
+ },
+ else => {
+ slot.state.store(cap_idle, .release);
+ return error.Unexpected;
+ },
+ }
+ const deadline = monotonicNs() + d.capture_timeout_ns;
+ while (true) {
+ const s = slot.state.load(.acquire);
+ switch (s) {
+ cap_captured => return,
+ cap_failed => {
+ slot.state.store(cap_idle, .release);
+ return error.Unsupported;
+ },
+ cap_armed => {
+ const now = monotonicNs();
+ if (now >= deadline) {
+ // Disarm; if the handler raced us it has moved on to
+ // `capturing` and we simply keep waiting for it.
+ if (slot.state.cmpxchgStrong(cap_armed, cap_idle, .acq_rel, .acquire) == null) return error.Timeout;
+ continue;
+ }
+ futexWaitNs(&slot.state, cap_armed, deadline - now);
+ },
+ // The handler is copying registers; it finishes promptly.
+ cap_capturing => futexWaitNs(&slot.state, cap_capturing, 1 * std.time.ns_per_ms),
+ else => unreachable,
+ }
+ }
+ }
+
+ /// Records the PT_LOAD ranges of every module `dl_iterate_phdr` reports
+ /// (for a static executable: the executable itself, not the vDSO).
+ fn scanModules(d: *Debug) void {
+ d.range_count = 0;
+ std.posix.dl_iterate_phdr(d, error{}, struct {
+ fn cb(info: *std.posix.dl_phdr_info, _: usize, ctx: *Debug) error{}!void {
+ for (info.phdr[0..info.phnum]) |phdr| {
+ if (phdr.type != .LOAD) continue;
+ if (ctx.range_count == max_ranges) return;
+ ctx.ranges[ctx.range_count] = .{ .start = info.addr +% phdr.vaddr, .len = phdr.memsz };
+ ctx.range_count += 1;
+ }
+ }
+ }.cb) catch {};
+ }
+
+ /// True when `addr` lies in a module `std.debug` already knows about, so
+ /// that asking it about `addr` cannot trigger a module rescan.
+ fn knownCode(d: *const Debug, addr: usize) bool {
+ for (d.ranges[0..d.range_count]) |r| {
+ if (addr >= r.start and addr - r.start < r.len) return true;
+ }
+ return false;
+ }
+
+ /// Unwinds from a saved context. A pc outside every known module (e.g. a
+ /// thread inside the vDSO) is reported as a single frame and not unwound,
+ /// because std would otherwise rescan its module list (see the header).
+ fn unwindContext(d: *const Debug, ctx: *const Native, addrs: *[max_frames]usize) std.debug.StackTrace {
+ if (!d.knownCode(ctx.getPc())) {
+ addrs[0] = ctx.getPc() +| 1;
+ return .{ .return_addresses = addrs[0..1], .skipped = .unknown };
+ }
+ return std.debug.captureCurrentStackTrace(.{ .context = ctx }, addrs);
+ }
+
+ /// True when nobody holds std.debug's `SelfInfo` lock right now. Called
+ /// with the target parked, so a held lock means the *target* (or another
+ /// live thread, which will let go) holds it; only the former deadlocks,
+ /// and the caller cannot tell them apart, so both yield `error.Busy`.
+ fn selfInfoFree(d: *const Debug) bool {
+ if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return true;
+ const di = std.debug.getSelfDebugInfo() catch return true;
+ if (!di.rwlock.tryLock(d.io)) return false;
+ di.rwlock.unlock(d.io);
+ return true;
+ }
+
+ fn releaseCapture(d: *Debug) void {
+ d.capture.state.store(cap_idle, .release);
+ futexWake(&d.capture.state);
+ }
+
+ fn writeFrames(d: *Debug, addrs: []const usize, w: *Writer) Error!void {
+ var fba = std.heap.FixedBufferAllocator.init(d.text_buf);
+ const alloc = fba.allocator();
+ const di = std.debug.getSelfDebugInfo() catch return error.Unsupported;
+ for (addrs, 0..) |ret_addr, i| {
+ // Return addresses point after the call; the first frame of a
+ // context capture is stored as pc+1 by std for the same reason.
+ const addr = ret_addr -| 1;
+ fba.reset();
+ var symbols: std.ArrayList(std.debug.Symbol) = .empty;
+ var sym = std.debug.Symbol.unknown;
+ if (d.knownCode(addr)) {
+ if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) {
+ if (symbols.items.len > 0) sym = symbols.items[0];
+ } else |_| {}
+ }
+ w.print("#{d} 0x{x} in {s} (", .{ i, addr, sym.name orelse "?" }) catch return error.WriteFailed;
+ if (sym.source_location) |sl| {
+ w.print("{s}:{d}:{d})\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed;
+ } else {
+ w.writeAll("?)\n") catch return error.WriteFailed;
+ }
+ }
+ }
+};
+
+// ------------------------------------------------------------------ handlers
+
+fn selfTid() u32 {
+ return @intCast(linux.gettid());
+}
+
+fn captureHandler(_: linux.SIG, _: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void {
+ const d = current orelse return;
+ const slot = &d.capture;
+ const me = selfTid();
+ if (slot.target_tid.load(.acquire) != me) return;
+ if (slot.state.cmpxchgStrong(cap_armed, cap_capturing, .acq_rel, .acquire) != null) return;
+ // The tid check and the swap are not one atomic step: a stale run (a
+ // signal that stayed pending while its request timed out) may have read
+ // the old tid and then won the swap of a request re-armed for another
+ // thread. `target_tid` is fixed while the slot is armed, so re-checking
+ // after the swap closes the window; hand the slot back untouched.
+ if (slot.target_tid.load(.acquire) != me) {
+ slot.state.store(cap_armed, .release);
+ futexWake(&slot.state);
+ return;
+ }
+ if (cpu_context.fromPosixSignalContext(ctx_ptr)) |ctx| {
+ slot.ctx = ctx;
+ slot.state.store(cap_captured, .release);
+ futexWake(&slot.state);
+ while (slot.state.load(.acquire) == cap_captured) futexWaitNs(&slot.state, cap_captured, null);
+ } else {
+ slot.state.store(cap_failed, .release);
+ futexWake(&slot.state);
+ }
+}
+
+/// aarch64 Linux ucontext_t, only as far as `mcontext.pc` (see
+/// std.debug.cpu_context's signal_ucontext_t).
+const UcontextAarch64 = extern struct {
+ flags: usize,
+ link: ?*UcontextAarch64,
+ stack: linux.stack_t,
+ sigmask: linux.sigset_t,
+ unused: [120]u8,
+ mcontext: extern struct {
+ fault_address: u64 align(16),
+ x: [30]u64,
+ lr: u64,
+ sp: u64,
+ pc: u64,
+ },
+};
+
+fn trapHandler(_: linux.SIG, info: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void {
+ const d = current orelse return trapFallback();
+ const ctx = cpu_context.fromPosixSignalContext(ctx_ptr) orelse return trapFallback();
+ // si_code > 0 is kernel-generated (TRAP_BRKPT for int3/brk); <= 0 is
+ // kill/tgkill/sigqueue from user space, where PC points at the
+ // interrupted instruction and must not be touched.
+ const from_instruction = info.code > 0;
+ if (comptime arch.isAARCH64()) {
+ // `brk #imm` does not advance PC; step over it so returning from the
+ // handler does not re-trap.
+ if (from_instruction) {
+ const uc: *UcontextAarch64 = @ptrCast(@alignCast(ctx_ptr.?));
+ uc.mcontext.pc += 4;
+ }
+ } else if (comptime arch != .x86_64) {
+ return trapFallback();
+ }
+ const tid = selfTid();
+ if (tid == server_tid.load(.acquire)) {
+ // Nobody could resume the thread that serves /breakpoints: step over.
+ _ = traps_skipped.fetchAdd(1, .acq_rel);
+ return;
+ }
+ const slot: *PauseSlot = for (d.paused[0..d.max_paused]) |*slot| {
+ if (slot.state.cmpxchgStrong(pause_free, pause_claimed, .acq_rel, .acquire) == null) break slot;
+ } else {
+ _ = traps_skipped.fetchAdd(1, .acq_rel);
+ return;
+ };
+ slot.ctx = ctx;
+ slot.tid.store(tid, .release);
+ slot.state.store(pause_paused, .release);
+ while (slot.state.load(.acquire) == pause_paused) futexWaitNs(&slot.state, pause_paused, null);
+ slot.state.store(pause_free, .release);
+}
+
+/// Restores the default SIGTRAP disposition and re-raises it: the signal is
+/// blocked while the handler runs, so it is delivered (fatally) on return.
+/// Only for a handler run with no `Debug` (a trap in flight during `deinit`)
+/// or on an architecture whose context cannot be read.
+fn trapFallback() void {
+ const act: linux.Sigaction = .{
+ .handler = .{ .handler = linux.SIG.DFL },
+ .mask = linux.sigemptyset(),
+ .flags = 0,
+ };
+ _ = linux.sigaction(.TRAP, &act, null);
+ _ = linux.tkill(linux.gettid(), .TRAP);
+}
+
+// --------------------------------------------------------------------- panic
+
+const panic_none: u32 = 0;
+const panic_recording: u32 = 1;
+const panic_recorded: u32 = 2;
+const panic_held: u32 = 3;
+const panic_continued: u32 = 4;
+
+var panic_state: std.atomic.Value(u32) = .init(panic_none);
+var panic_msg: [panic_msg_cap]u8 = undefined;
+var panic_msg_len: usize = 0;
+var panic_addrs: [max_frames]usize = undefined;
+var panic_addr_count: usize = 0;
+/// The tid of the panicking thread (0 before any panic).
+pub var panic_tid: u32 = 0;
+
+/// Records the first panic: message (bounded copy) and stack addresses.
+/// Returns false if a panic was already recorded (nested or second panic).
+pub fn recordPanic(msg: []const u8, first_trace_addr: ?usize) bool {
+ if (panic_state.cmpxchgStrong(panic_none, panic_recording, .acq_rel, .acquire) != null) return false;
+ panic_tid = selfTid();
+ panic_msg_len = @min(msg.len, panic_msg.len);
+ @memcpy(panic_msg[0..panic_msg_len], msg[0..panic_msg_len]);
+ const trace = std.debug.captureCurrentStackTrace(.{ .first_address = first_trace_addr }, &panic_addrs);
+ panic_addr_count = trace.return_addresses.len;
+ panic_state.store(panic_recorded, .release);
+ return true;
+}
+
+/// Parks the panicking thread until `Debug.panicContinue` when holding is
+/// enabled and a `Debug` exists; then hands over to `std.debug.defaultPanic`.
+pub fn panicHook(msg: []const u8, first_trace_addr: ?usize) noreturn {
+ @branchHint(.cold);
+ if (recordPanic(msg, first_trace_addr)) {
+ // The server thread cannot be held: it is the one that would have to
+ // serve /panic/ctl.
+ if (hold_on_panic and current != null and panic_tid != server_tid.load(.acquire)) {
+ if (panic_state.cmpxchgStrong(panic_recorded, panic_held, .acq_rel, .acquire) == null) {
+ while (panic_state.load(.acquire) == panic_held) futexWaitNs(&panic_state, panic_held, null);
+ }
+ }
+ }
+ std.debug.defaultPanic(msg, first_trace_addr);
+}
+
+/// Clears the recorded panic. Only meaningful in tests of the record path.
+pub fn resetPanicRecord() void {
+ panic_msg_len = 0;
+ panic_addr_count = 0;
+ panic_tid = 0;
+ panic_state.store(panic_none, .release);
+}
+
+// ------------------------------------------------------------------- helpers
+
+fn futexWake(word: *std.atomic.Value(u32)) void {
+ _ = linux.futex_3arg(&word.raw, .{ .cmd = .WAKE, .private = true }, std.math.maxInt(u32));
+}
+
+/// Waits while `*word == expect`, at most `timeout_ns` (forever when null).
+/// Returns on wake, timeout, value change or EINTR; callers loop.
+fn futexWaitNs(word: *std.atomic.Value(u32), expect: u32, timeout_ns: ?u64) void {
+ var ts: linux.timespec = undefined;
+ const ts_ptr: ?*const linux.timespec = if (timeout_ns) |ns| blk: {
+ ts = .{ .sec = @intCast(ns / std.time.ns_per_s), .nsec = @intCast(ns % std.time.ns_per_s) };
+ break :blk &ts;
+ } else null;
+ _ = linux.futex_4arg(&word.raw, .{ .cmd = .WAIT, .private = true }, expect, ts_ptr);
+}
+
+fn monotonicNs() u64 {
+ var ts: linux.timespec = undefined;
+ _ = linux.clock_gettime(.MONOTONIC, &ts);
+ return @as(u64, @intCast(ts.sec)) * std.time.ns_per_s + @as(u64, @intCast(ts.nsec));
+}
+
+const FileError = error{ NotFound, Unexpected, TooBig };
+
+/// Reads a whole (small) file with raw syscalls.
+fn readFile(path: [*:0]const u8, buf: []u8) FileError![]u8 {
+ const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0);
+ switch (linux.errno(fd_rc)) {
+ .SUCCESS => {},
+ .NOENT, .SRCH => return error.NotFound,
+ else => return error.Unexpected,
+ }
+ const fd: i32 = @intCast(fd_rc);
+ defer _ = linux.close(fd);
+ var len: usize = 0;
+ while (len < buf.len) {
+ const rc = linux.read(fd, buf[len..].ptr, buf.len - len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR => continue,
+ .SRCH, .NOENT => return error.NotFound,
+ else => return error.Unexpected,
+ }
+ if (rc == 0) return buf[0..len];
+ len += rc;
+ }
+ return error.TooBig;
+}
+
+/// One pread of `buf.len` bytes at `offset`; 0 at the end of the file.
+fn preadFile(path: [*:0]const u8, offset: u64, buf: []u8) Error!usize {
+ const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0);
+ if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected;
+ const fd: i32 = @intCast(fd_rc);
+ defer _ = linux.close(fd);
+ var len: usize = 0;
+ while (len < buf.len) {
+ const rc = linux.pread(fd, buf[len..].ptr, buf.len - len, @intCast(offset + len));
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR => continue,
+ else => return error.Unexpected,
+ }
+ if (rc == 0) break;
+ len += rc;
+ }
+ return len;
+}
+
+/// Streams a file of any size to `w`.
+fn streamFile(path: [*:0]const u8, w: *Writer) Error!void {
+ const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0);
+ if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected;
+ const fd: i32 = @intCast(fd_rc);
+ defer _ = linux.close(fd);
+ var buf: [4096]u8 = undefined;
+ while (true) {
+ const rc = linux.read(fd, &buf, buf.len);
+ switch (linux.errno(rc)) {
+ .SUCCESS => {},
+ .INTR => continue,
+ else => return error.Unexpected,
+ }
+ if (rc == 0) return;
+ w.writeAll(buf[0..rc]) catch return error.WriteFailed;
+ }
+}
+
+fn writeHexLines(base: usize, bytes: []const u8, w: *Writer) Error!void {
+ var offset: usize = 0;
+ while (offset < bytes.len) : (offset += 16) {
+ const line = bytes[offset..@min(offset + 16, bytes.len)];
+ w.print("{x:0>[1]} ", .{ base +% offset, @sizeOf(usize) * 2 }) catch return error.WriteFailed;
+ for (line, 0..) |byte, i| {
+ w.print("{X:0>2} ", .{byte}) catch return error.WriteFailed;
+ if (i == 7) w.writeByte(' ') catch return error.WriteFailed;
+ }
+ w.writeByte(' ') catch return error.WriteFailed;
+ if (line.len < 16) {
+ var missing = (16 - line.len) * 3;
+ if (line.len < 8) missing += 1;
+ w.splatByteAll(' ', missing) catch return error.WriteFailed;
+ }
+ for (line) |byte| {
+ w.writeByte(if (std.ascii.isPrint(byte)) byte else '.') catch return error.WriteFailed;
+ }
+ w.writeByte('\n') catch return error.WriteFailed;
+ }
+}
+
+fn writeRegs(ctx: *const Native, w: *Writer) Error!void {
+ if (comptime arch == .x86_64) {
+ inline for (@typeInfo(Native.Gpr).@"enum".fields) |f| {
+ w.print("{s} 0x{x}\n", .{ f.name, ctx.gprs.get(@field(Native.Gpr, f.name)) }) catch return error.WriteFailed;
+ }
+ w.print("pc 0x{x}\nsp 0x{x}\nfp 0x{x}\n", .{
+ ctx.gprs.get(.rip), ctx.gprs.get(.rsp), ctx.gprs.get(.rbp),
+ }) catch return error.WriteFailed;
+ } else if (comptime arch.isAARCH64()) {
+ for (ctx.x, 0..) |x, i| w.print("x{d} 0x{x}\n", .{ i, x }) catch return error.WriteFailed;
+ w.print("sp 0x{x}\npc 0x{x}\nfp 0x{x}\nlr 0x{x}\n", .{
+ ctx.sp, ctx.pc, ctx.x[29], ctx.x[30],
+ }) catch return error.WriteFailed;
+ } else {
+ w.print("pc 0x{x}\nfp 0x{x}\n", .{ ctx.getPc(), ctx.getFp() }) catch return error.WriteFailed;
+ }
+}
+
+// --------------------------------------------------------------------- tests
+
+const testing = std.testing;
+
+fn testOptions(text_buf: []u8) Options {
+ return .{ .io = testing.io, .text_buf = text_buf };
+}
+
+noinline fn sleepMs(ms: u64) void {
+ var ts: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * std.time.ns_per_ms) };
+ _ = linux.nanosleep(&ts, null);
+}
+
+// The test threads use atomic builtins rather than `std.atomic.Value` methods
+// so that, in release modes, their pc is never inside an inlined callee: the
+// DWARF symbolizer names the innermost inlined function at an address (see
+// the notes on `writeFrames`).
+const SpinState = struct {
+ tid: std.atomic.Value(u32) = .init(0),
+ stop: bool = false,
+ counter: u32 = 0,
+ done: bool = false,
+};
+
+noinline fn spinHere(st: *SpinState) void {
+ while (!@atomicLoad(bool, &st.stop, .acquire)) {
+ _ = @atomicRmw(u32, &st.counter, .Add, 1, .monotonic);
+ }
+}
+
+fn spinThreadMain(st: *SpinState) void {
+ st.tid.store(selfTid(), .release);
+ spinHere(st);
+ @atomicStore(bool, &st.done, true, .release); // keeps the call above from becoming a tail call
+}
+
+fn waitForTid(st: *SpinState) u32 {
+ var tries: usize = 0;
+ while (st.tid.load(.acquire) == 0) : (tries += 1) {
+ if (tries > 2000) return 0;
+ sleepMs(1);
+ }
+ return st.tid.load(.acquire);
+}
+
+test "capture own stack" {
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+ try testing.expect(current == &d);
+
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ try d.threadStack(selfTid(), &out.writer);
+ const text = out.written();
+ try testing.expect(std.mem.indexOf(u8, text, "#0 0x") != null);
+ try testing.expect(std.mem.indexOf(u8, text, "debug.zig:") != null);
+ try testing.expect(std.mem.indexOf(u8, text, "test.capture own stack") != null);
+
+ out.clearRetainingCapacity();
+ try d.threadRegs(selfTid(), &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null);
+}
+
+test "capture another thread: stack, regs, name, stat" {
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+
+ var st: SpinState = .{};
+ const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st});
+ const tid = waitForTid(&st);
+ try testing.expect(tid != 0);
+
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ try d.threadStack(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "spinThreadMain") != null);
+
+ out.clearRetainingCapacity();
+ try d.threadRegs(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null);
+
+ out.clearRetainingCapacity();
+ try d.threadName(tid, &out.writer);
+ try testing.expect(out.written().len > 0);
+ try testing.expect(std.mem.indexOfScalar(u8, out.written(), '\n') == null);
+
+ out.clearRetainingCapacity();
+ try d.threadStat(tid, &out.writer);
+ try testing.expect(std.mem.startsWith(u8, out.written(), "state "));
+ try testing.expect(std.mem.indexOf(u8, out.written(), "\nutime ") != null);
+
+ // Enumeration lists both threads and nothing bogus.
+ try testing.expect(d.threadExists(tid));
+ try testing.expect(d.threadExists(selfTid()));
+ var found_self = false;
+ var found_other = false;
+ var i: usize = 0;
+ var prev: u32 = 0;
+ while (d.threadAt(i)) |t| : (i += 1) {
+ try testing.expect(t > prev);
+ prev = t;
+ if (t == tid) found_other = true;
+ if (t == selfTid()) found_self = true;
+ }
+ try testing.expect(found_self and found_other);
+
+ // Repeated captures of the same thread keep working.
+ var k: usize = 0;
+ while (k < 5) : (k += 1) {
+ out.clearRetainingCapacity();
+ try d.threadStack(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null);
+ }
+ const before = @atomicLoad(u32, &st.counter, .acquire);
+ sleepMs(2);
+ try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != before); // the thread is running again
+
+ @atomicStore(bool, &st.stop, true, .release);
+ th.join();
+ try testing.expect(!d.threadExists(tid));
+ try testing.expectError(error.NoThread, d.threadStack(tid, &out.writer));
+ try testing.expectError(error.NoThread, d.threadName(tid, &out.writer));
+}
+
+/// The address of the call site in the caller, i.e. inside this file's test.
+noinline fn callerAddress() usize {
+ return @returnAddress() - 1;
+}
+
+test "resolveAddr names this file" {
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ try d.resolveAddr(callerAddress(), &out.writer);
+ const text = out.written();
+ var lines = std.mem.splitScalar(u8, text, '\n');
+ const fn_name = lines.next().?;
+ const loc = lines.next().?;
+ const module = lines.next().?;
+ try testing.expect(fn_name.len > 0 and !std.mem.eql(u8, fn_name, "?"));
+ try testing.expect(std.mem.indexOf(u8, loc, "debug.zig:") != null);
+ try testing.expect(module.len > 0);
+
+ out.clearRetainingCapacity();
+ try d.resolveAddr(8, &out.writer);
+ try testing.expectEqualStrings("?\n?\n?\n", out.written());
+
+ // Regression: an unmapped lookup must not poison std's unwind cache (see
+ // the header); unwinding afterwards still works.
+ out.clearRetainingCapacity();
+ try d.threadStack(selfTid(), &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "test.resolveAddr names this file") != null);
+}
+
+test "readMem, writeMem, hexdump" {
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+
+ var value: [8]u8 = .{ 1, 2, 3, 4, 5, 6, 7, 8 };
+ var got: [8]u8 = undefined;
+ try testing.expectEqual(@as(usize, 8), try d.readMem(@intFromPtr(&value), &got));
+ try testing.expectEqualSlices(u8, &value, &got);
+ try testing.expectError(error.Unmapped, d.readMem(8, &got));
+
+ const new = [_]u8{ 0xaa, 0xbb, 0xcc };
+ try testing.expectEqual(@as(usize, 3), try d.writeMem(@intFromPtr(&value) + 2, &new));
+ try testing.expectEqualSlices(u8, &.{ 1, 2, 0xaa, 0xbb, 0xcc, 6, 7, 8 }, &value);
+ try testing.expectError(error.Unmapped, d.writeMem(8, &new));
+
+ var bytes: [19]u8 = .{ 0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff, 0x01, 0x12, 0x13 };
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ try d.hexdump(@intFromPtr(&bytes), bytes.len, &out.writer);
+ const expected = try std.fmt.allocPrint(testing.allocator,
+ \\{x:0>[2]} 00 11 22 33 44 55 66 77 88 99 AA BB CC DD EE FF .."3DUfw........
+ \\{x:0>[2]} 01 12 13 ...
+ \\
+ , .{ @intFromPtr(&bytes), @intFromPtr(&bytes) + 16, @sizeOf(usize) * 2 });
+ defer testing.allocator.free(expected);
+ try testing.expectEqualStrings(expected, out.written());
+ try testing.expectError(error.Unmapped, d.hexdump(8, 16, &out.writer));
+
+ // Address 0 (also reached by an offset that wraps) must be an error, not a
+ // safety-checked null pointer cast on the server thread.
+ try testing.expectError(error.Unmapped, d.readMem(0, &got));
+ try testing.expectError(error.Unmapped, d.writeMem(0, &new));
+ try testing.expectError(error.Unmapped, d.hexdump(0, 16, &out.writer));
+ try testing.expectError(error.Unmapped, d.readMem(std.math.maxInt(usize) - 3, &got));
+ try testing.expectError(error.Unmapped, d.hexdump(std.math.maxInt(usize) - 3, 16, &out.writer));
+
+ out.clearRetainingCapacity();
+ try d.maps(&out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "[stack]") != null);
+
+ // readMaps serves the file piecewise at any offset and ends with 0.
+ var piece: [4096]u8 = undefined;
+ var total: usize = 0;
+ while (true) {
+ const n = try d.readMaps(total, &piece);
+ if (n == 0) break;
+ total += n;
+ }
+ try testing.expect(total >= out.written().len / 2);
+ try testing.expectEqual(@as(usize, 0), try d.readMaps(std.math.maxInt(u64), &piece));
+}
+
+test "breakpoint on the server thread and past the slot table steps over; tgkill SIGTRAP parks" {
+ if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest;
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ var opts = testOptions(&text_buf);
+ opts.max_paused = 1;
+ try d.init(opts);
+ defer d.deinit();
+ try d.enableBreakpoints();
+ defer d.disableBreakpoints();
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+
+ // The "server" thread (this one, for the test) hits a breakpoint: it keeps running.
+ const skipped0 = traps_skipped.load(.acquire);
+ server_tid.store(selfTid(), .release);
+ defer server_tid.store(0, .release);
+ @breakpoint();
+ try testing.expectEqual(skipped0 + 1, traps_skipped.load(.acquire));
+ try testing.expect(!d.isPaused(selfTid()));
+
+ // One slot: the first trapping thread parks, the second steps over.
+ var a: TrapState = .{};
+ const ta = try std.Thread.spawn(.{}, trapThreadMain, .{&a});
+ var tries: usize = 0;
+ while (a.tid.load(.acquire) == 0 or !d.isPaused(a.tid.load(.acquire))) : (tries += 1) {
+ try testing.expect(tries < 5000);
+ sleepMs(1);
+ }
+ var b: TrapState = .{};
+ const tb = try std.Thread.spawn(.{}, trapThreadMain, .{&b});
+ tb.join();
+ try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &b.counter, .acquire));
+ try testing.expectEqual(skipped0 + 2, traps_skipped.load(.acquire));
+ try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &a.counter, .acquire));
+ try d.resumeThread(a.tid.load(.acquire));
+ ta.join();
+ try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &a.counter, .acquire));
+
+ // A SIGTRAP sent with tgkill (not an int3/brk) parks the thread where it
+ // was; resuming it must not skip an instruction: the spinner keeps counting.
+ var st: SpinState = .{};
+ const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st});
+ const tid = waitForTid(&st);
+ try testing.expect(tid != 0);
+ try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.tgkill(linux.getpid(), @intCast(tid), .TRAP)));
+ tries = 0;
+ while (!d.isPaused(tid)) : (tries += 1) {
+ try testing.expect(tries < 5000);
+ sleepMs(1);
+ }
+ const frozen = @atomicLoad(u32, &st.counter, .acquire);
+ sleepMs(5);
+ try testing.expectEqual(frozen, @atomicLoad(u32, &st.counter, .acquire));
+ out.clearRetainingCapacity();
+ try d.pausedStack(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null);
+ try d.resumeThread(tid);
+ sleepMs(5);
+ try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != frozen);
+ @atomicStore(bool, &st.stop, true, .release);
+ th.join();
+}
+
+const LockState = struct {
+ tid: std.atomic.Value(u32) = .init(0),
+ release: std.atomic.Value(bool) = .init(false),
+ unlocked: std.atomic.Value(bool) = .init(false),
+ stop: std.atomic.Value(bool) = .init(false),
+ io: std.Io,
+};
+
+fn lockHolderMain(st: *LockState) void {
+ const di = std.debug.getSelfDebugInfo() catch return;
+ di.rwlock.lockUncancelable(st.io);
+ st.tid.store(selfTid(), .release);
+ while (!st.release.load(.acquire)) sleepMs(1);
+ di.rwlock.unlock(st.io);
+ st.unlocked.store(true, .release);
+ while (!st.stop.load(.acquire)) sleepMs(1);
+}
+
+test "a target parked while holding std.debug's lock is Busy, not a deadlock" {
+ if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return error.SkipZigTest;
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+ var st: LockState = .{ .io = testing.io };
+ const th = try std.Thread.spawn(.{}, lockHolderMain, .{&st});
+ var tries: usize = 0;
+ while (st.tid.load(.acquire) == 0) : (tries += 1) {
+ try testing.expect(tries < 2000);
+ sleepMs(1);
+ }
+ const tid = st.tid.load(.acquire);
+ // No allocation while the holder has the lock: `testing.allocator`
+ // records a stack trace per allocation, which needs that same lock.
+ var buf: [16 * 1024]u8 = undefined;
+ var w: Writer = .fixed(&buf);
+ try testing.expectError(error.Busy, d.threadStack(tid, &w));
+ try testing.expectEqual(cap_idle, d.capture.state.load(.acquire));
+ // Registers need no unwind and are still available.
+ try d.threadRegs(tid, &w);
+ try testing.expect(std.mem.indexOf(u8, w.buffered(), "pc 0x") != null);
+ // Handshake, not a sleep: a slow holder would otherwise still hold the
+ // lock and the next capture would legitimately be Busy again.
+ st.release.store(true, .release);
+ tries = 0;
+ while (!st.unlocked.load(.acquire)) : (tries += 1) {
+ try testing.expect(tries < 5000);
+ sleepMs(1);
+ }
+ w = .fixed(&buf);
+ try d.threadStack(tid, &w);
+ try testing.expect(std.mem.indexOf(u8, w.buffered(), "lockHolderMain") != null);
+ st.stop.store(true, .release);
+ th.join();
+}
+
+const TrapState = struct {
+ tid: std.atomic.Value(u32) = .init(0),
+ counter: u32 = 0,
+};
+
+noinline fn trapThreadMain(st: *TrapState) void {
+ st.tid.store(selfTid(), .release);
+ @breakpoint();
+ _ = @atomicRmw(u32, &st.counter, .Add, 1, .acq_rel);
+}
+
+test "breakpoint: pause, inspect, resume" {
+ if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest;
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+ try d.enableBreakpoints();
+
+ var st: TrapState = .{};
+ const th = try std.Thread.spawn(.{}, trapThreadMain, .{&st});
+ var tries: usize = 0;
+ while (st.tid.load(.acquire) == 0 or !d.isPaused(st.tid.load(.acquire))) : (tries += 1) {
+ try testing.expect(tries < 5000);
+ sleepMs(1);
+ }
+ const tid = st.tid.load(.acquire);
+ try testing.expectEqual(@as(?u32, tid), d.pausedAt(0));
+ try testing.expectEqual(@as(?u32, null), d.pausedAt(1));
+ try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire));
+
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ try d.pausedStack(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "trapThreadMain") != null);
+ out.clearRetainingCapacity();
+ try d.pausedRegs(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null);
+
+ // A paused thread can also be captured through the signal path.
+ out.clearRetainingCapacity();
+ try d.threadStack(tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null);
+
+ sleepMs(5);
+ try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire));
+ try d.resumeThread(tid);
+ th.join();
+ try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &st.counter, .acquire));
+ try testing.expect(!d.isPaused(tid));
+ try testing.expectEqual(@as(?u32, null), d.pausedAt(0));
+ try testing.expectError(error.NotPaused, d.resumeThread(tid));
+ try testing.expectError(error.NotPaused, d.pausedStack(tid, &out.writer));
+ d.disableBreakpoints();
+}
+
+/// Stands in for `FullPanic`'s call: the first trace address is the return
+/// address into the panicking function.
+noinline fn panicLike(msg: []const u8) bool {
+ return recordPanic(msg, @returnAddress());
+}
+
+test "panic record path" {
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+ defer resetPanicRecord();
+
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ try d.panicMessage(&out.writer);
+ try testing.expectEqualStrings("", out.written());
+ try testing.expect(!d.panicHeld());
+ try testing.expectError(error.NoPanic, d.panicContinue());
+
+ try testing.expect(panicLike("something broke"));
+ try testing.expect(!recordPanic("nested", null));
+ try testing.expectEqual(selfTid(), panic_tid);
+
+ try d.panicMessage(&out.writer);
+ try testing.expectEqualStrings("something broke", out.written());
+ out.clearRetainingCapacity();
+ try d.panicStack(&out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "test.panic record path") != null);
+ try testing.expect(!d.panicHeld());
+ try testing.expectError(error.NoPanic, d.panicContinue());
+
+ // A long message is truncated, not overflowed.
+ resetPanicRecord();
+ const long = [_]u8{'x'} ** (panic_msg_cap + 100);
+ try testing.expect(recordPanic(&long, null));
+ out.clearRetainingCapacity();
+ try d.panicMessage(&out.writer);
+ try testing.expectEqual(@as(usize, panic_msg_cap), out.written().len);
+}
+
+const MaskState = struct {
+ tid: std.atomic.Value(u32) = .init(0),
+ unblock: std.atomic.Value(bool) = .init(false),
+ stop: std.atomic.Value(bool) = .init(false),
+ signal: linux.SIG,
+};
+
+fn maskedThreadMain(st: *MaskState) void {
+ var set = linux.sigemptyset();
+ linux.sigaddset(&set, st.signal);
+ _ = linux.sigprocmask(linux.SIG.BLOCK, &set, null);
+ st.tid.store(selfTid(), .release);
+ while (!st.unblock.load(.acquire)) sleepMs(1);
+ _ = linux.sigprocmask(linux.SIG.UNBLOCK, &set, null);
+ while (!st.stop.load(.acquire)) sleepMs(1);
+}
+
+test "capture timeout on a thread with the signal masked" {
+ var text_buf: [16 * 1024]u8 = undefined;
+ var d: Debug = undefined;
+ var opts = testOptions(&text_buf);
+ opts.capture_timeout_ns = 50 * std.time.ns_per_ms;
+ try d.init(opts);
+ defer d.deinit();
+
+ var st: MaskState = .{ .signal = d.capture_signal };
+ const th = try std.Thread.spawn(.{}, maskedThreadMain, .{&st});
+ var tries: usize = 0;
+ while (st.tid.load(.acquire) == 0) : (tries += 1) {
+ try testing.expect(tries < 2000);
+ sleepMs(1);
+ }
+ const masked_tid = st.tid.load(.acquire);
+
+ var out: Writer.Allocating = .init(testing.allocator);
+ defer out.deinit();
+ const t0 = monotonicNs();
+ try testing.expectError(error.Timeout, d.threadStack(masked_tid, &out.writer));
+ try testing.expect(monotonicNs() - t0 >= 50 * std.time.ns_per_ms);
+ try testing.expectEqual(cap_idle, d.capture.state.load(.acquire));
+
+ // The process is healthy: another thread can still be captured...
+ var spin: SpinState = .{};
+ const spinner = try std.Thread.spawn(.{}, spinThreadMain, .{&spin});
+ const spin_tid = waitForTid(&spin);
+ try testing.expect(spin_tid != 0);
+ out.clearRetainingCapacity();
+ try d.threadStack(spin_tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null);
+
+ // ...and the late delivery of the pending signal is harmless.
+ st.unblock.store(true, .release);
+ sleepMs(20);
+ out.clearRetainingCapacity();
+ try d.threadStack(spin_tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null);
+ out.clearRetainingCapacity();
+ try d.threadStack(masked_tid, &out.writer);
+ try testing.expect(std.mem.indexOf(u8, out.written(), "maskedThreadMain") != null);
+
+ @atomicStore(bool, &spin.stop, true, .release);
+ spinner.join();
+ st.stop.store(true, .release);
+ th.join();
+}
+
+test "options validation and single instance" {
+ var text_buf: [4096]u8 = undefined;
+ var d: Debug = undefined;
+ var opts = testOptions(&text_buf);
+ opts.capture_signal = 5;
+ try testing.expectError(error.InvalidOptions, d.init(opts));
+ opts = testOptions(&text_buf);
+ opts.max_paused = max_paused_cap + 1;
+ try testing.expectError(error.InvalidOptions, d.init(opts));
+ try d.init(testOptions(&text_buf));
+ defer d.deinit();
+ var d2: Debug = undefined;
+ try testing.expectError(error.AlreadyInitialized, d2.init(testOptions(&text_buf)));
+}