From ba996acfcad1698adbf4a1834fe50e73b1c6cab9 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Sat, 19 Sep 2026 23:28:22 -0300 Subject: Rename programs: 9player -> 9ns, introspect -> 9proc, app -> web (9web) Directories, binaries, build options (-D9ns, -D9proc), step names, module name (9proc), thread and fs names, env var NINEPLAYER_MOUNT -> NINE_MOUNT, docs and test scripts. Browser assets move to web/static. Co-Authored-By: Claude Fable 5.1 --- 9proc/src/linux/debug.zig | 1458 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 1458 insertions(+) create mode 100644 9proc/src/linux/debug.zig (limited to '9proc/src/linux/debug.zig') diff --git a/9proc/src/linux/debug.zig b/9proc/src/linux/debug.zig new file mode 100644 index 0000000..a344380 --- /dev/null +++ b/9proc/src/linux/debug.zig @@ -0,0 +1,1458 @@ +//! Linux debug facilities for the 9proc server: threads, stacks, +//! registers, address → source, memory, breakpoints and panics. +//! +//! This file is a pure API; a later adapter turns it into a core `Provider`. +//! All text is written to a `*std.Io.Writer`. Nothing here allocates after +//! `init` except from the caller-provided `text_buf`, which is used as a fixed +//! arena for symbol text and reset before every query. +//! +//! Only one `Debug` may exist per process: the signal handlers and the panic +//! hook find their state through the global `current` pointer set by `init`. +//! +//! Mechanics +//! +//! * Capturing another thread's stack or registers: the calling (server) +//! thread sends `capture_signal` with `tgkill`. The SA_SIGINFO handler copies +//! the interrupted register state (`cpu_context.fromPosixSignalContext`) into +//! the single capture slot and parks on a futex. The server unwinds the +//! parked thread's stack from that context, releases the target, then +//! symbolizes. The handler is async-signal-safe: no allocation, no +//! `std.debug`, no locks other than the futex. A target that does not run +//! the handler within `capture_timeout_ns` (signal masked, thread in D +//! state, ...) yields `error.Timeout`; a late-arriving handler run cannot +//! corrupt a reused slot because it must match the requested tid and win a +//! compare-and-swap from `armed` on the slot state (that pair plays the role +//! of a generation counter: a stale run finds the slot idle, armed for +//! another tid, or armed for itself, in which case its capture is simply the +//! valid answer to the new request). +//! * Breakpoints: `@breakpoint()` raises SIGTRAP on the executing thread only. +//! The handler claims a pause slot, saves the context and parks on a futex +//! until `resumeThread`. On x86_64 the saved PC is already past `int3`; on +//! aarch64 the handler advances PC by 4 in the ucontext before returning +//! (only for a real `brk`, i.e. a kernel-generated si_code; a SIGTRAP sent +//! with kill/tgkill parks the thread where it was). With no free slot the +//! thread steps over the breakpoint and keeps running (`traps_skipped` +//! counts them): the debug layer never kills the process. The server thread +//! itself (`server_tid`) is never parked, a breakpoint there is stepped +//! over, because nobody could resume it. Only a stale handler run after +//! `deinit` (no `current`) falls back to the default disposition. +//! * Panics: `panicHook` records the message and a stack capture, then, if +//! `hold_on_panic` and a `Debug` exists, parks until `panicContinue`; then +//! `std.debug.defaultPanic` runs. A nested or second panic, or a panic on +//! the server thread itself (which could never be continued), goes +//! straight to the default handler. +//! * std.debug's `SelfInfo` guards its state with an `Io.RwLock`. A target +//! parked while holding it (a thread inside a stack-trace dump, say) would +//! deadlock the unwind, so after parking a thread the lock is probed with +//! `tryLock`; a held lock yields `error.Busy` and the target is released. +//! * Known-module guard: `std.debug.SelfInfo` (Zig 0.16) rebuilds its module +//! list whenever it is asked about an address outside every known module, +//! freeing the CIE lists its unwind cache still points into; later unwinds +//! then read freed memory. `init` records the PT_LOAD ranges of the +//! executable (the same source std uses) and every lookup or unwind is +//! first checked against them; addresses outside (unmapped, vDSO, ...) +//! render as "?" and are never handed to std. + +const std = @import("std"); +const builtin = @import("builtin"); +const linux = std.os.linux; +const cpu_context = std.debug.cpu_context; +const Writer = std.Io.Writer; +const Native = cpu_context.Native; +const arch = builtin.cpu.arch; + +pub const Options = struct { + /// Used for `std.debug` symbolization (reading debug info from disk). + io: std.Io, + /// Fixed arena for symbol text. A `FixedBufferAllocator` is placed over it + /// and reset before every query. 16 KiB is plenty; 4 KiB is a sane floor. + text_buf: []u8, + /// Real-time signal used to snapshot other threads. SIGRTMIN is 32 on + /// Linux without libc; the default is SIGRTMIN+3. + capture_signal: u8 = default_capture_signal, + /// How long to wait for a target thread to run the capture handler. + capture_timeout_ns: u64 = 250 * std.time.ns_per_ms, + /// How many threads may be parked in `@breakpoint()` at once (≤ 32). + max_paused: u8 = 16, +}; + +pub const default_capture_signal: u8 = 32 + 3; + +/// Hard upper bound of `Options.max_paused` (slot storage is static). +pub const max_paused_cap = 32; +/// Maximum number of frames written by any stack function. +pub const max_frames = 64; +/// Maximum number of tids enumerated from /proc/self/task. +pub const max_threads = 512; +/// Upper bound of the recorded panic message. +pub const panic_msg_cap = 1024; +/// Maximum number of PT_LOAD ranges recorded by the known-module guard. +pub const max_ranges = 64; + +/// Consulted by `panicHook`: when true and a `Debug` is initialized, the +/// panicking thread is held until `panicContinue`. +pub var hold_on_panic: bool = true; + +/// The one live instance, set by `init`, cleared by `deinit`. +pub var current: ?*Debug = null; + +/// The tid of the thread serving requests (0 = none). That thread is never +/// parked by a breakpoint or held by a panic, since nobody could release it. +pub var server_tid: std.atomic.Value(u32) = .init(0); + +/// Breakpoints stepped over because no pause slot was free, or because they +/// were hit on the server thread. +pub var traps_skipped: std.atomic.Value(u32) = .init(0); + +pub const Error = error{ + /// The target thread did not run the capture handler in time. + Timeout, + /// No thread with that tid exists in this process. + NoThread, + /// The address is not mapped (EFAULT from process_vm_readv/writev). + Unmapped, + /// The thread is not parked in a breakpoint. + NotPaused, + /// No panic has been recorded / is being held. + NoPanic, + /// Another `Debug` already exists in this process. + AlreadyInitialized, + /// The operation is not available on this architecture / kernel. + Unsupported, + /// The target thread is parked inside std.debug (holding its lock); its + /// stack cannot be unwound without deadlocking. Retry later. + Busy, + /// Invalid option value. + InvalidOptions, + /// A syscall or /proc read failed unexpectedly. + Unexpected, + /// The writer failed. + WriteFailed, +}; + +// Capture slot states. +const cap_idle: u32 = 0; +const cap_armed: u32 = 1; +const cap_capturing: u32 = 2; +const cap_captured: u32 = 3; +const cap_failed: u32 = 4; + +// Pause slot states. +const pause_free: u32 = 0; +const pause_claimed: u32 = 1; +const pause_paused: u32 = 2; +const pause_resuming: u32 = 3; + +const CaptureSlot = struct { + state: std.atomic.Value(u32) = .init(cap_idle), + target_tid: std.atomic.Value(u32) = .init(0), + ctx: Native = undefined, +}; + +const PauseSlot = struct { + state: std.atomic.Value(u32) = .init(pause_free), + tid: std.atomic.Value(u32) = .init(0), + ctx: Native = undefined, +}; + +pub const Debug = struct { + io: std.Io, + text_buf: []u8, + capture_signal: linux.SIG, + capture_timeout_ns: u64, + max_paused: u8, + + capture: CaptureSlot = .{}, + paused: [max_paused_cap]PauseSlot = [_]PauseSlot{.{}} ** max_paused_cap, + + old_capture_action: linux.Sigaction = undefined, + old_trap_action: linux.Sigaction = undefined, + breakpoints_enabled: bool = false, + + tids: [max_threads]u32 = undefined, + tid_count: usize = 0, + + ranges: [max_ranges]Range = undefined, + range_count: usize = 0, + + const Range = struct { start: usize, len: usize }; + + /// Installs the capture handler (not the SIGTRAP handler) and publishes + /// `d` as `current`. + pub fn init(d: *Debug, opts: Options) Error!void { + if (current != null) return error.AlreadyInitialized; + if (opts.capture_signal < 32 or opts.capture_signal >= linux.NSIG) return error.InvalidOptions; + if (opts.max_paused == 0 or opts.max_paused > max_paused_cap) return error.InvalidOptions; + if (Native == noreturn) return error.Unsupported; + d.* = .{ + .io = opts.io, + .text_buf = opts.text_buf, + .capture_signal = @enumFromInt(opts.capture_signal), + .capture_timeout_ns = opts.capture_timeout_ns, + .max_paused = opts.max_paused, + }; + d.scanModules(); + const act: linux.Sigaction = .{ + .handler = .{ .sigaction = captureHandler }, + .mask = linux.sigemptyset(), + .flags = linux.SA.SIGINFO | linux.SA.RESTART, + }; + current = d; + if (linux.errno(linux.sigaction(d.capture_signal, &act, &d.old_capture_action)) != .SUCCESS) { + current = null; + return error.Unexpected; + } + } + + /// Restores the signal dispositions and clears `current`. Threads parked + /// in a breakpoint are resumed first. + pub fn deinit(d: *Debug) void { + d.disableBreakpoints(); + _ = linux.sigaction(d.capture_signal, &d.old_capture_action, null); + if (current == d) current = null; + } + + /// Installs the SIGTRAP handler so that `@breakpoint()` parks the thread. + pub fn enableBreakpoints(d: *Debug) Error!void { + if (d.breakpoints_enabled) return; + if (arch != .x86_64 and !arch.isAARCH64()) return error.Unsupported; + const act: linux.Sigaction = .{ + .handler = .{ .sigaction = trapHandler }, + .mask = linux.sigemptyset(), + .flags = linux.SA.SIGINFO | linux.SA.RESTART, + }; + if (linux.errno(linux.sigaction(.TRAP, &act, &d.old_trap_action)) != .SUCCESS) return error.Unexpected; + d.breakpoints_enabled = true; + } + + /// Restores the previous SIGTRAP disposition and resumes every parked thread. + pub fn disableBreakpoints(d: *Debug) void { + if (!d.breakpoints_enabled) return; + _ = linux.sigaction(.TRAP, &d.old_trap_action, null); + d.breakpoints_enabled = false; + for (&d.paused) |*slot| { + if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) == null) + futexWake(&slot.state); + } + } + + // ---------------------------------------------------------------- threads + + /// The nth tid of this process, numerically sorted; null past the end. + /// Index 0 rescans /proc/self/task; higher indices reuse that scan. + pub fn threadAt(d: *Debug, index: usize) ?u32 { + if (index == 0 or d.tid_count == 0) d.scanThreads(); + if (index >= d.tid_count) return null; + return d.tids[index]; + } + + pub fn threadExists(d: *Debug, tid: u32) bool { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return false; + var buf: [32]u8 = undefined; + _ = readFile(path, &buf) catch return false; + return true; + } + + /// The thread's comm (without the trailing newline). + pub fn threadName(d: *Debug, tid: u32, w: *Writer) Error!void { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/comm", .{tid}) catch return error.Unexpected; + var buf: [64]u8 = undefined; + const text = readFile(path, &buf) catch |err| switch (err) { + error.NotFound => return error.NoThread, + else => return error.Unexpected, + }; + w.writeAll(std.mem.trimEnd(u8, text, "\n")) catch return error.WriteFailed; + } + + /// A few fields of /proc/self/task//stat, one "name value" per line: + /// state, utime, stime, minflt, majflt, priority, nice, processor. + pub fn threadStat(d: *Debug, tid: u32, w: *Writer) Error!void { + _ = d; + var path_buf: [64]u8 = undefined; + const path = std.fmt.bufPrintZ(&path_buf, "/proc/self/task/{d}/stat", .{tid}) catch return error.Unexpected; + var buf: [1024]u8 = undefined; + const text = readFile(path, &buf) catch |err| switch (err) { + error.NotFound => return error.NoThread, + else => return error.Unexpected, + }; + // " () S "; comm may contain spaces and parens. + const close = std.mem.lastIndexOfScalar(u8, text, ')') orelse return error.Unexpected; + var it = std.mem.tokenizeScalar(u8, text[close + 1 ..], ' '); + // Field numbers below are 0-based from `state`. + const wanted = [_]struct { idx: usize, name: []const u8 }{ + .{ .idx = 0, .name = "state" }, + .{ .idx = 11, .name = "utime" }, + .{ .idx = 12, .name = "stime" }, + .{ .idx = 7, .name = "minflt" }, + .{ .idx = 9, .name = "majflt" }, + .{ .idx = 15, .name = "priority" }, + .{ .idx = 16, .name = "nice" }, + .{ .idx = 36, .name = "processor" }, + }; + var fields: [40][]const u8 = undefined; + var n: usize = 0; + while (it.next()) |f| : (n += 1) { + if (n == fields.len) break; + fields[n] = f; + } + for (wanted) |want| { + const value = if (want.idx < n) fields[want.idx] else "?"; + w.print("{s} {s}\n", .{ want.name, value }) catch return error.WriteFailed; + } + } + + /// "#n 0x in (::)" per frame. The calling + /// thread unwinds itself directly; any other thread is captured with the + /// capture signal. + pub fn threadStack(d: *Debug, tid: u32, w: *Writer) Error!void { + var addrs: [max_frames]usize = undefined; + var trace: std.debug.StackTrace = undefined; + if (tid == selfTid()) { + trace = std.debug.captureCurrentStackTrace(.{}, &addrs); + } else { + try d.captureThread(tid); + if (!d.selfInfoFree()) { + d.releaseCapture(); + return error.Busy; + } + trace = d.unwindContext(&d.capture.ctx, &addrs); + d.releaseCapture(); + } + try d.writeFrames(trace.return_addresses, w); + } + + /// " 0x" per general register, plus pc/sp/fp aliases. + pub fn threadRegs(d: *Debug, tid: u32, w: *Writer) Error!void { + if (tid == selfTid()) { + const ctx = Native.current(); + return writeRegs(&ctx, w); + } + try d.captureThread(tid); + const ctx = d.capture.ctx; + d.releaseCapture(); + return writeRegs(&ctx, w); + } + + // ------------------------------------------------------ addresses & memory + + /// "fn\nfile:line:col\nmodule\n", unknown parts as "?". + pub fn resolveAddr(d: *Debug, addr: usize, w: *Writer) Error!void { + if (!d.knownCode(addr)) return w.writeAll("?\n?\n?\n") catch error.WriteFailed; + var fba = std.heap.FixedBufferAllocator.init(d.text_buf); + const alloc = fba.allocator(); + const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; + var sym = std.debug.Symbol.unknown; + var symbols: std.ArrayList(std.debug.Symbol) = .empty; + if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { + if (symbols.items.len > 0) sym = symbols.items[0]; + } else |_| {} + w.print("{s}\n", .{sym.name orelse "?"}) catch return error.WriteFailed; + if (sym.source_location) |sl| { + w.print("{s}:{d}:{d}\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; + } else { + w.writeAll("?\n") catch return error.WriteFailed; + } + const module = di.getModuleName(d.io, addr) catch "?"; + w.print("{s}\n", .{module}) catch return error.WriteFailed; + } + + /// Reads `buf.len` bytes at `addr` via process_vm_readv on the own + /// process. Never faults. Returns the number of bytes read (short when the + /// range crosses into an unmapped page); `error.Unmapped` when nothing + /// could be read. + pub fn readMem(d: *Debug, addr: usize, buf: []u8) Error!usize { + _ = d; + if (buf.len == 0) return 0; + // Page 0 is never mapped (mmap_min_addr) and a null `iovec.base` is a + // safety-checked cast; the same answer without the trap. + if (addr == 0) return error.Unmapped; + const local = [_]std.posix.iovec{.{ .base = buf.ptr, .len = buf.len }}; + const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = buf.len }}; + const rc = linux.process_vm_readv(linux.getpid(), &local, &remote, 0); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .FAULT => return error.Unmapped, + .NOSYS, .PERM => return error.Unsupported, + else => return error.Unexpected, + } + } + + /// Writes `data` at `addr` via process_vm_writev. Read-only mappings also + /// report `error.Unmapped` (the kernel says EFAULT for both). + pub fn writeMem(d: *Debug, addr: usize, data: []const u8) Error!usize { + _ = d; + if (data.len == 0) return 0; + if (addr == 0) return error.Unmapped; + const local = [_]std.posix.iovec_const{.{ .base = data.ptr, .len = data.len }}; + const remote = [_]std.posix.iovec_const{.{ .base = @ptrFromInt(addr), .len = data.len }}; + const rc = linux.process_vm_writev(linux.getpid(), &local, &remote, 0); + switch (linux.errno(rc)) { + .SUCCESS => return rc, + .FAULT => return error.Unmapped, + .NOSYS, .PERM => return error.Unsupported, + else => return error.Unexpected, + } + } + + /// Hexdump of `len` bytes at `addr` in the shape of `std.debug.dumpHex` + /// (16 bytes per line, address column, bytes in two groups, ASCII column). + /// Stops early at the first unmapped byte; `error.Unmapped` only when the + /// very first chunk is unreadable. + pub fn hexdump(d: *Debug, addr: usize, len: usize, w: *Writer) Error!void { + var chunk: [256]u8 = undefined; + var done: usize = 0; + while (done < len) { + const want = @min(chunk.len, len - done); + const got = d.readMem(addr +% done, chunk[0..want]) catch |err| switch (err) { + error.Unmapped => if (done == 0) return error.Unmapped else break, + else => return err, + }; + if (got == 0) break; + try writeHexLines(addr +% done, chunk[0..got], w); + done += got; + if (got < want) break; + } + } + + /// Copies /proc/self/maps to `w`. + pub fn maps(d: *Debug, w: *Writer) Error!void { + _ = d; + return streamFile("/proc/self/maps", w); + } + + /// Reads `buf.len` bytes of /proc/self/maps at `offset` (0 at the end). + /// Not a consistent snapshot across reads; a map appearing between two + /// reads shifts the text, like `cat` on /proc itself. + pub fn readMaps(d: *Debug, offset: u64, buf: []u8) Error!usize { + _ = d; + if (offset > std.math.maxInt(i64)) return 0; + return preadFile("/proc/self/maps", offset, buf); + } + + // ------------------------------------------------------------ breakpoints + + /// The nth tid currently parked in `@breakpoint()`. + pub fn pausedAt(d: *Debug, index: usize) ?u32 { + var n: usize = 0; + for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.load(.acquire) != pause_paused) continue; + if (n == index) return slot.tid.load(.acquire); + n += 1; + } + return null; + } + + pub fn isPaused(d: *Debug, tid: u32) bool { + return d.pausedSlot(tid) != null; + } + + pub fn pausedStack(d: *Debug, tid: u32, w: *Writer) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + if (!d.selfInfoFree()) return error.Busy; + var addrs: [max_frames]usize = undefined; + const trace = d.unwindContext(&slot.ctx, &addrs); + try d.writeFrames(trace.return_addresses, w); + } + + pub fn pausedRegs(d: *Debug, tid: u32, w: *Writer) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + return writeRegs(&slot.ctx, w); + } + + /// Lets a parked thread continue past its breakpoint. + pub fn resumeThread(d: *Debug, tid: u32) Error!void { + const slot = d.pausedSlot(tid) orelse return error.NotPaused; + if (slot.state.cmpxchgStrong(pause_paused, pause_resuming, .acq_rel, .acquire) != null) return error.NotPaused; + futexWake(&slot.state); + } + + fn pausedSlot(d: *Debug, tid: u32) ?*PauseSlot { + for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.load(.acquire) == pause_paused and slot.tid.load(.acquire) == tid) return slot; + } + return null; + } + + // ------------------------------------------------------------------ panic + + /// The recorded panic message; nothing before any panic. + pub fn panicMessage(d: *Debug, w: *Writer) Error!void { + _ = d; + if (panic_state.load(.acquire) == panic_none) return; + w.writeAll(panic_msg[0..panic_msg_len]) catch return error.WriteFailed; + } + + /// Frames of the panicking thread, symbolized lazily. + pub fn panicStack(d: *Debug, w: *Writer) Error!void { + if (panic_state.load(.acquire) == panic_none) return; + try d.writeFrames(panic_addrs[0..panic_addr_count], w); + } + + /// True while a panicking thread is parked waiting for `panicContinue`. + pub fn panicHeld(d: *Debug) bool { + _ = d; + return panic_state.load(.acquire) == panic_held; + } + + /// Releases the held panicking thread into `std.debug.defaultPanic`. + pub fn panicContinue(d: *Debug) Error!void { + _ = d; + if (panic_state.cmpxchgStrong(panic_held, panic_continued, .acq_rel, .acquire) != null) return error.NoPanic; + futexWake(&panic_state); + } + + // -------------------------------------------------------------- internals + + fn scanThreads(d: *Debug) void { + d.tid_count = 0; + const fd_rc = linux.open("/proc/self/task", .{ .ACCMODE = .RDONLY, .DIRECTORY = true, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var buf: [4096]u8 align(@alignOf(linux.dirent64)) = undefined; + while (true) { + const rc = linux.getdents64(fd, &buf, buf.len); + if (linux.errno(rc) != .SUCCESS or rc == 0) break; + var off: usize = 0; + while (off < rc) { + const ent: *align(1) const linux.dirent64 = @ptrCast(&buf[off]); + const name_ptr: [*:0]const u8 = @ptrCast(&buf[off + @offsetOf(linux.dirent64, "name")]); + const name = std.mem.span(name_ptr); + if (std.fmt.parseInt(u32, name, 10)) |tid| { + if (d.tid_count < max_threads) { + d.tids[d.tid_count] = tid; + d.tid_count += 1; + } + } else |_| {} + off += ent.reclen; + } + } + std.mem.sort(u32, d.tids[0..d.tid_count], {}, std.sort.asc(u32)); + } + + /// Arms the capture slot for `tid`, signals it and waits until the handler + /// has parked with its context copied. On success the caller owns the + /// slot until `releaseCapture`. + fn captureThread(d: *Debug, tid: u32) Error!void { + const slot = &d.capture; + slot.target_tid.store(tid, .release); + slot.state.store(cap_armed, .release); + const rc = linux.tgkill(linux.getpid(), @intCast(tid), d.capture_signal); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .SRCH => { + slot.state.store(cap_idle, .release); + return error.NoThread; + }, + else => { + slot.state.store(cap_idle, .release); + return error.Unexpected; + }, + } + const deadline = monotonicNs() + d.capture_timeout_ns; + while (true) { + const s = slot.state.load(.acquire); + switch (s) { + cap_captured => return, + cap_failed => { + slot.state.store(cap_idle, .release); + return error.Unsupported; + }, + cap_armed => { + const now = monotonicNs(); + if (now >= deadline) { + // Disarm; if the handler raced us it has moved on to + // `capturing` and we simply keep waiting for it. + if (slot.state.cmpxchgStrong(cap_armed, cap_idle, .acq_rel, .acquire) == null) return error.Timeout; + continue; + } + futexWaitNs(&slot.state, cap_armed, deadline - now); + }, + // The handler is copying registers; it finishes promptly. + cap_capturing => futexWaitNs(&slot.state, cap_capturing, 1 * std.time.ns_per_ms), + else => unreachable, + } + } + } + + /// Records the PT_LOAD ranges of every module `dl_iterate_phdr` reports + /// (for a static executable: the executable itself, not the vDSO). + fn scanModules(d: *Debug) void { + d.range_count = 0; + std.posix.dl_iterate_phdr(d, error{}, struct { + fn cb(info: *std.posix.dl_phdr_info, _: usize, ctx: *Debug) error{}!void { + for (info.phdr[0..info.phnum]) |phdr| { + if (phdr.type != .LOAD) continue; + if (ctx.range_count == max_ranges) return; + ctx.ranges[ctx.range_count] = .{ .start = info.addr +% phdr.vaddr, .len = phdr.memsz }; + ctx.range_count += 1; + } + } + }.cb) catch {}; + } + + /// True when `addr` lies in a module `std.debug` already knows about, so + /// that asking it about `addr` cannot trigger a module rescan. + fn knownCode(d: *const Debug, addr: usize) bool { + for (d.ranges[0..d.range_count]) |r| { + if (addr >= r.start and addr - r.start < r.len) return true; + } + return false; + } + + /// Unwinds from a saved context. A pc outside every known module (e.g. a + /// thread inside the vDSO) is reported as a single frame and not unwound, + /// because std would otherwise rescan its module list (see the header). + fn unwindContext(d: *const Debug, ctx: *const Native, addrs: *[max_frames]usize) std.debug.StackTrace { + if (!d.knownCode(ctx.getPc())) { + addrs[0] = ctx.getPc() +| 1; + return .{ .return_addresses = addrs[0..1], .skipped = .unknown }; + } + return std.debug.captureCurrentStackTrace(.{ .context = ctx }, addrs); + } + + /// True when nobody holds std.debug's `SelfInfo` lock right now. Called + /// with the target parked, so a held lock means the *target* (or another + /// live thread, which will let go) holds it; only the former deadlocks, + /// and the caller cannot tell them apart, so both yield `error.Busy`. + fn selfInfoFree(d: *const Debug) bool { + if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return true; + const di = std.debug.getSelfDebugInfo() catch return true; + if (!di.rwlock.tryLock(d.io)) return false; + di.rwlock.unlock(d.io); + return true; + } + + fn releaseCapture(d: *Debug) void { + d.capture.state.store(cap_idle, .release); + futexWake(&d.capture.state); + } + + fn writeFrames(d: *Debug, addrs: []const usize, w: *Writer) Error!void { + var fba = std.heap.FixedBufferAllocator.init(d.text_buf); + const alloc = fba.allocator(); + const di = std.debug.getSelfDebugInfo() catch return error.Unsupported; + for (addrs, 0..) |ret_addr, i| { + // Return addresses point after the call; the first frame of a + // context capture is stored as pc+1 by std for the same reason. + const addr = ret_addr -| 1; + fba.reset(); + var symbols: std.ArrayList(std.debug.Symbol) = .empty; + var sym = std.debug.Symbol.unknown; + if (d.knownCode(addr)) { + if (di.getSymbols(d.io, alloc, alloc, addr, true, &symbols)) { + if (symbols.items.len > 0) sym = symbols.items[0]; + } else |_| {} + } + w.print("#{d} 0x{x} in {s} (", .{ i, addr, sym.name orelse "?" }) catch return error.WriteFailed; + if (sym.source_location) |sl| { + w.print("{s}:{d}:{d})\n", .{ sl.file_name, sl.line, sl.column }) catch return error.WriteFailed; + } else { + w.writeAll("?)\n") catch return error.WriteFailed; + } + } + } +}; + +// ------------------------------------------------------------------ handlers + +fn selfTid() u32 { + return @intCast(linux.gettid()); +} + +fn captureHandler(_: linux.SIG, _: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { + const d = current orelse return; + const slot = &d.capture; + const me = selfTid(); + if (slot.target_tid.load(.acquire) != me) return; + if (slot.state.cmpxchgStrong(cap_armed, cap_capturing, .acq_rel, .acquire) != null) return; + // The tid check and the swap are not one atomic step: a stale run (a + // signal that stayed pending while its request timed out) may have read + // the old tid and then won the swap of a request re-armed for another + // thread. `target_tid` is fixed while the slot is armed, so re-checking + // after the swap closes the window; hand the slot back untouched. + if (slot.target_tid.load(.acquire) != me) { + slot.state.store(cap_armed, .release); + futexWake(&slot.state); + return; + } + if (cpu_context.fromPosixSignalContext(ctx_ptr)) |ctx| { + slot.ctx = ctx; + slot.state.store(cap_captured, .release); + futexWake(&slot.state); + while (slot.state.load(.acquire) == cap_captured) futexWaitNs(&slot.state, cap_captured, null); + } else { + slot.state.store(cap_failed, .release); + futexWake(&slot.state); + } +} + +/// aarch64 Linux ucontext_t, only as far as `mcontext.pc` (see +/// std.debug.cpu_context's signal_ucontext_t). +const UcontextAarch64 = extern struct { + flags: usize, + link: ?*UcontextAarch64, + stack: linux.stack_t, + sigmask: linux.sigset_t, + unused: [120]u8, + mcontext: extern struct { + fault_address: u64 align(16), + x: [30]u64, + lr: u64, + sp: u64, + pc: u64, + }, +}; + +fn trapHandler(_: linux.SIG, info: *const linux.siginfo_t, ctx_ptr: ?*anyopaque) callconv(.c) void { + const d = current orelse return trapFallback(); + const ctx = cpu_context.fromPosixSignalContext(ctx_ptr) orelse return trapFallback(); + // si_code > 0 is kernel-generated (TRAP_BRKPT for int3/brk); <= 0 is + // kill/tgkill/sigqueue from user space, where PC points at the + // interrupted instruction and must not be touched. + const from_instruction = info.code > 0; + if (comptime arch.isAARCH64()) { + // `brk #imm` does not advance PC; step over it so returning from the + // handler does not re-trap. + if (from_instruction) { + const uc: *UcontextAarch64 = @ptrCast(@alignCast(ctx_ptr.?)); + uc.mcontext.pc += 4; + } + } else if (comptime arch != .x86_64) { + return trapFallback(); + } + const tid = selfTid(); + if (tid == server_tid.load(.acquire)) { + // Nobody could resume the thread that serves /breakpoints: step over. + _ = traps_skipped.fetchAdd(1, .acq_rel); + return; + } + const slot: *PauseSlot = for (d.paused[0..d.max_paused]) |*slot| { + if (slot.state.cmpxchgStrong(pause_free, pause_claimed, .acq_rel, .acquire) == null) break slot; + } else { + _ = traps_skipped.fetchAdd(1, .acq_rel); + return; + }; + slot.ctx = ctx; + slot.tid.store(tid, .release); + slot.state.store(pause_paused, .release); + while (slot.state.load(.acquire) == pause_paused) futexWaitNs(&slot.state, pause_paused, null); + slot.state.store(pause_free, .release); +} + +/// Restores the default SIGTRAP disposition and re-raises it: the signal is +/// blocked while the handler runs, so it is delivered (fatally) on return. +/// Only for a handler run with no `Debug` (a trap in flight during `deinit`) +/// or on an architecture whose context cannot be read. +fn trapFallback() void { + const act: linux.Sigaction = .{ + .handler = .{ .handler = linux.SIG.DFL }, + .mask = linux.sigemptyset(), + .flags = 0, + }; + _ = linux.sigaction(.TRAP, &act, null); + _ = linux.tkill(linux.gettid(), .TRAP); +} + +// --------------------------------------------------------------------- panic + +const panic_none: u32 = 0; +const panic_recording: u32 = 1; +const panic_recorded: u32 = 2; +const panic_held: u32 = 3; +const panic_continued: u32 = 4; + +var panic_state: std.atomic.Value(u32) = .init(panic_none); +var panic_msg: [panic_msg_cap]u8 = undefined; +var panic_msg_len: usize = 0; +var panic_addrs: [max_frames]usize = undefined; +var panic_addr_count: usize = 0; +/// The tid of the panicking thread (0 before any panic). +pub var panic_tid: u32 = 0; + +/// Records the first panic: message (bounded copy) and stack addresses. +/// Returns false if a panic was already recorded (nested or second panic). +pub fn recordPanic(msg: []const u8, first_trace_addr: ?usize) bool { + if (panic_state.cmpxchgStrong(panic_none, panic_recording, .acq_rel, .acquire) != null) return false; + panic_tid = selfTid(); + panic_msg_len = @min(msg.len, panic_msg.len); + @memcpy(panic_msg[0..panic_msg_len], msg[0..panic_msg_len]); + const trace = std.debug.captureCurrentStackTrace(.{ .first_address = first_trace_addr }, &panic_addrs); + panic_addr_count = trace.return_addresses.len; + panic_state.store(panic_recorded, .release); + return true; +} + +/// Parks the panicking thread until `Debug.panicContinue` when holding is +/// enabled and a `Debug` exists; then hands over to `std.debug.defaultPanic`. +pub fn panicHook(msg: []const u8, first_trace_addr: ?usize) noreturn { + @branchHint(.cold); + if (recordPanic(msg, first_trace_addr)) { + // The server thread cannot be held: it is the one that would have to + // serve /panic/ctl. + if (hold_on_panic and current != null and panic_tid != server_tid.load(.acquire)) { + if (panic_state.cmpxchgStrong(panic_recorded, panic_held, .acq_rel, .acquire) == null) { + while (panic_state.load(.acquire) == panic_held) futexWaitNs(&panic_state, panic_held, null); + } + } + } + std.debug.defaultPanic(msg, first_trace_addr); +} + +/// Clears the recorded panic. Only meaningful in tests of the record path. +pub fn resetPanicRecord() void { + panic_msg_len = 0; + panic_addr_count = 0; + panic_tid = 0; + panic_state.store(panic_none, .release); +} + +// ------------------------------------------------------------------- helpers + +fn futexWake(word: *std.atomic.Value(u32)) void { + _ = linux.futex_3arg(&word.raw, .{ .cmd = .WAKE, .private = true }, std.math.maxInt(u32)); +} + +/// Waits while `*word == expect`, at most `timeout_ns` (forever when null). +/// Returns on wake, timeout, value change or EINTR; callers loop. +fn futexWaitNs(word: *std.atomic.Value(u32), expect: u32, timeout_ns: ?u64) void { + var ts: linux.timespec = undefined; + const ts_ptr: ?*const linux.timespec = if (timeout_ns) |ns| blk: { + ts = .{ .sec = @intCast(ns / std.time.ns_per_s), .nsec = @intCast(ns % std.time.ns_per_s) }; + break :blk &ts; + } else null; + _ = linux.futex_4arg(&word.raw, .{ .cmd = .WAIT, .private = true }, expect, ts_ptr); +} + +fn monotonicNs() u64 { + var ts: linux.timespec = undefined; + _ = linux.clock_gettime(.MONOTONIC, &ts); + return @as(u64, @intCast(ts.sec)) * std.time.ns_per_s + @as(u64, @intCast(ts.nsec)); +} + +const FileError = error{ NotFound, Unexpected, TooBig }; + +/// Reads a whole (small) file with raw syscalls. +fn readFile(path: [*:0]const u8, buf: []u8) FileError![]u8 { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + switch (linux.errno(fd_rc)) { + .SUCCESS => {}, + .NOENT, .SRCH => return error.NotFound, + else => return error.Unexpected, + } + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var len: usize = 0; + while (len < buf.len) { + const rc = linux.read(fd, buf[len..].ptr, buf.len - len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + .SRCH, .NOENT => return error.NotFound, + else => return error.Unexpected, + } + if (rc == 0) return buf[0..len]; + len += rc; + } + return error.TooBig; +} + +/// One pread of `buf.len` bytes at `offset`; 0 at the end of the file. +fn preadFile(path: [*:0]const u8, offset: u64, buf: []u8) Error!usize { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var len: usize = 0; + while (len < buf.len) { + const rc = linux.pread(fd, buf[len..].ptr, buf.len - len, @intCast(offset + len)); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => return error.Unexpected, + } + if (rc == 0) break; + len += rc; + } + return len; +} + +/// Streams a file of any size to `w`. +fn streamFile(path: [*:0]const u8, w: *Writer) Error!void { + const fd_rc = linux.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true }, 0); + if (linux.errno(fd_rc) != .SUCCESS) return error.Unexpected; + const fd: i32 = @intCast(fd_rc); + defer _ = linux.close(fd); + var buf: [4096]u8 = undefined; + while (true) { + const rc = linux.read(fd, &buf, buf.len); + switch (linux.errno(rc)) { + .SUCCESS => {}, + .INTR => continue, + else => return error.Unexpected, + } + if (rc == 0) return; + w.writeAll(buf[0..rc]) catch return error.WriteFailed; + } +} + +fn writeHexLines(base: usize, bytes: []const u8, w: *Writer) Error!void { + var offset: usize = 0; + while (offset < bytes.len) : (offset += 16) { + const line = bytes[offset..@min(offset + 16, bytes.len)]; + w.print("{x:0>[1]} ", .{ base +% offset, @sizeOf(usize) * 2 }) catch return error.WriteFailed; + for (line, 0..) |byte, i| { + w.print("{X:0>2} ", .{byte}) catch return error.WriteFailed; + if (i == 7) w.writeByte(' ') catch return error.WriteFailed; + } + w.writeByte(' ') catch return error.WriteFailed; + if (line.len < 16) { + var missing = (16 - line.len) * 3; + if (line.len < 8) missing += 1; + w.splatByteAll(' ', missing) catch return error.WriteFailed; + } + for (line) |byte| { + w.writeByte(if (std.ascii.isPrint(byte)) byte else '.') catch return error.WriteFailed; + } + w.writeByte('\n') catch return error.WriteFailed; + } +} + +fn writeRegs(ctx: *const Native, w: *Writer) Error!void { + if (comptime arch == .x86_64) { + inline for (@typeInfo(Native.Gpr).@"enum".fields) |f| { + w.print("{s} 0x{x}\n", .{ f.name, ctx.gprs.get(@field(Native.Gpr, f.name)) }) catch return error.WriteFailed; + } + w.print("pc 0x{x}\nsp 0x{x}\nfp 0x{x}\n", .{ + ctx.gprs.get(.rip), ctx.gprs.get(.rsp), ctx.gprs.get(.rbp), + }) catch return error.WriteFailed; + } else if (comptime arch.isAARCH64()) { + for (ctx.x, 0..) |x, i| w.print("x{d} 0x{x}\n", .{ i, x }) catch return error.WriteFailed; + w.print("sp 0x{x}\npc 0x{x}\nfp 0x{x}\nlr 0x{x}\n", .{ + ctx.sp, ctx.pc, ctx.x[29], ctx.x[30], + }) catch return error.WriteFailed; + } else { + w.print("pc 0x{x}\nfp 0x{x}\n", .{ ctx.getPc(), ctx.getFp() }) catch return error.WriteFailed; + } +} + +// --------------------------------------------------------------------- tests + +const testing = std.testing; + +fn testOptions(text_buf: []u8) Options { + return .{ .io = testing.io, .text_buf = text_buf }; +} + +noinline fn sleepMs(ms: u64) void { + var ts: linux.timespec = .{ .sec = @intCast(ms / 1000), .nsec = @intCast((ms % 1000) * std.time.ns_per_ms) }; + _ = linux.nanosleep(&ts, null); +} + +// The test threads use atomic builtins rather than `std.atomic.Value` methods +// so that, in release modes, their pc is never inside an inlined callee: the +// DWARF symbolizer names the innermost inlined function at an address (see +// the notes on `writeFrames`). +const SpinState = struct { + tid: std.atomic.Value(u32) = .init(0), + stop: bool = false, + counter: u32 = 0, + done: bool = false, +}; + +noinline fn spinHere(st: *SpinState) void { + while (!@atomicLoad(bool, &st.stop, .acquire)) { + _ = @atomicRmw(u32, &st.counter, .Add, 1, .monotonic); + } +} + +fn spinThreadMain(st: *SpinState) void { + st.tid.store(selfTid(), .release); + spinHere(st); + @atomicStore(bool, &st.done, true, .release); // keeps the call above from becoming a tail call +} + +fn waitForTid(st: *SpinState) u32 { + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + if (tries > 2000) return 0; + sleepMs(1); + } + return st.tid.load(.acquire); +} + +test "capture own stack" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + try testing.expect(current == &d); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.threadStack(selfTid(), &out.writer); + const text = out.written(); + try testing.expect(std.mem.indexOf(u8, text, "#0 0x") != null); + try testing.expect(std.mem.indexOf(u8, text, "debug.zig:") != null); + try testing.expect(std.mem.indexOf(u8, text, "test.capture own stack") != null); + + out.clearRetainingCapacity(); + try d.threadRegs(selfTid(), &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); +} + +test "capture another thread: stack, regs, name, stat" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + + var st: SpinState = .{}; + const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); + const tid = waitForTid(&st); + try testing.expect(tid != 0); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinThreadMain") != null); + + out.clearRetainingCapacity(); + try d.threadRegs(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x0\n") == null); + + out.clearRetainingCapacity(); + try d.threadName(tid, &out.writer); + try testing.expect(out.written().len > 0); + try testing.expect(std.mem.indexOfScalar(u8, out.written(), '\n') == null); + + out.clearRetainingCapacity(); + try d.threadStat(tid, &out.writer); + try testing.expect(std.mem.startsWith(u8, out.written(), "state ")); + try testing.expect(std.mem.indexOf(u8, out.written(), "\nutime ") != null); + + // Enumeration lists both threads and nothing bogus. + try testing.expect(d.threadExists(tid)); + try testing.expect(d.threadExists(selfTid())); + var found_self = false; + var found_other = false; + var i: usize = 0; + var prev: u32 = 0; + while (d.threadAt(i)) |t| : (i += 1) { + try testing.expect(t > prev); + prev = t; + if (t == tid) found_other = true; + if (t == selfTid()) found_self = true; + } + try testing.expect(found_self and found_other); + + // Repeated captures of the same thread keep working. + var k: usize = 0; + while (k < 5) : (k += 1) { + out.clearRetainingCapacity(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + } + const before = @atomicLoad(u32, &st.counter, .acquire); + sleepMs(2); + try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != before); // the thread is running again + + @atomicStore(bool, &st.stop, true, .release); + th.join(); + try testing.expect(!d.threadExists(tid)); + try testing.expectError(error.NoThread, d.threadStack(tid, &out.writer)); + try testing.expectError(error.NoThread, d.threadName(tid, &out.writer)); +} + +/// The address of the call site in the caller, i.e. inside this file's test. +noinline fn callerAddress() usize { + return @returnAddress() - 1; +} + +test "resolveAddr names this file" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.resolveAddr(callerAddress(), &out.writer); + const text = out.written(); + var lines = std.mem.splitScalar(u8, text, '\n'); + const fn_name = lines.next().?; + const loc = lines.next().?; + const module = lines.next().?; + try testing.expect(fn_name.len > 0 and !std.mem.eql(u8, fn_name, "?")); + try testing.expect(std.mem.indexOf(u8, loc, "debug.zig:") != null); + try testing.expect(module.len > 0); + + out.clearRetainingCapacity(); + try d.resolveAddr(8, &out.writer); + try testing.expectEqualStrings("?\n?\n?\n", out.written()); + + // Regression: an unmapped lookup must not poison std's unwind cache (see + // the header); unwinding afterwards still works. + out.clearRetainingCapacity(); + try d.threadStack(selfTid(), &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "test.resolveAddr names this file") != null); +} + +test "readMem, writeMem, hexdump" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + + var value: [8]u8 = .{ 1, 2, 3, 4, 5, 6, 7, 8 }; + var got: [8]u8 = undefined; + try testing.expectEqual(@as(usize, 8), try d.readMem(@intFromPtr(&value), &got)); + try testing.expectEqualSlices(u8, &value, &got); + try testing.expectError(error.Unmapped, d.readMem(8, &got)); + + const new = [_]u8{ 0xaa, 0xbb, 0xcc }; + try testing.expectEqual(@as(usize, 3), try d.writeMem(@intFromPtr(&value) + 2, &new)); + try testing.expectEqualSlices(u8, &.{ 1, 2, 0xaa, 0xbb, 0xcc, 6, 7, 8 }, &value); + try testing.expectError(error.Unmapped, d.writeMem(8, &new)); + + var bytes: [19]u8 = .{ 0x00, 0x11, 0x22, 0x33, 0x44, 0x55, 0x66, 0x77, 0x88, 0x99, 0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff, 0x01, 0x12, 0x13 }; + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.hexdump(@intFromPtr(&bytes), bytes.len, &out.writer); + const expected = try std.fmt.allocPrint(testing.allocator, + \\{x:0>[2]} 00 11 22 33 44 55 66 77 88 99 AA BB CC DD EE FF .."3DUfw........ + \\{x:0>[2]} 01 12 13 ... + \\ + , .{ @intFromPtr(&bytes), @intFromPtr(&bytes) + 16, @sizeOf(usize) * 2 }); + defer testing.allocator.free(expected); + try testing.expectEqualStrings(expected, out.written()); + try testing.expectError(error.Unmapped, d.hexdump(8, 16, &out.writer)); + + // Address 0 (also reached by an offset that wraps) must be an error, not a + // safety-checked null pointer cast on the server thread. + try testing.expectError(error.Unmapped, d.readMem(0, &got)); + try testing.expectError(error.Unmapped, d.writeMem(0, &new)); + try testing.expectError(error.Unmapped, d.hexdump(0, 16, &out.writer)); + try testing.expectError(error.Unmapped, d.readMem(std.math.maxInt(usize) - 3, &got)); + try testing.expectError(error.Unmapped, d.hexdump(std.math.maxInt(usize) - 3, 16, &out.writer)); + + out.clearRetainingCapacity(); + try d.maps(&out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "[stack]") != null); + + // readMaps serves the file piecewise at any offset and ends with 0. + var piece: [4096]u8 = undefined; + var total: usize = 0; + while (true) { + const n = try d.readMaps(total, &piece); + if (n == 0) break; + total += n; + } + try testing.expect(total >= out.written().len / 2); + try testing.expectEqual(@as(usize, 0), try d.readMaps(std.math.maxInt(u64), &piece)); +} + +test "breakpoint on the server thread and past the slot table steps over; tgkill SIGTRAP parks" { + if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.max_paused = 1; + try d.init(opts); + defer d.deinit(); + try d.enableBreakpoints(); + defer d.disableBreakpoints(); + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + + // The "server" thread (this one, for the test) hits a breakpoint: it keeps running. + const skipped0 = traps_skipped.load(.acquire); + server_tid.store(selfTid(), .release); + defer server_tid.store(0, .release); + @breakpoint(); + try testing.expectEqual(skipped0 + 1, traps_skipped.load(.acquire)); + try testing.expect(!d.isPaused(selfTid())); + + // One slot: the first trapping thread parks, the second steps over. + var a: TrapState = .{}; + const ta = try std.Thread.spawn(.{}, trapThreadMain, .{&a}); + var tries: usize = 0; + while (a.tid.load(.acquire) == 0 or !d.isPaused(a.tid.load(.acquire))) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + var b: TrapState = .{}; + const tb = try std.Thread.spawn(.{}, trapThreadMain, .{&b}); + tb.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &b.counter, .acquire)); + try testing.expectEqual(skipped0 + 2, traps_skipped.load(.acquire)); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &a.counter, .acquire)); + try d.resumeThread(a.tid.load(.acquire)); + ta.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &a.counter, .acquire)); + + // A SIGTRAP sent with tgkill (not an int3/brk) parks the thread where it + // was; resuming it must not skip an instruction: the spinner keeps counting. + var st: SpinState = .{}; + const th = try std.Thread.spawn(.{}, spinThreadMain, .{&st}); + const tid = waitForTid(&st); + try testing.expect(tid != 0); + try testing.expectEqual(linux.E.SUCCESS, linux.errno(linux.tgkill(linux.getpid(), @intCast(tid), .TRAP))); + tries = 0; + while (!d.isPaused(tid)) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + const frozen = @atomicLoad(u32, &st.counter, .acquire); + sleepMs(5); + try testing.expectEqual(frozen, @atomicLoad(u32, &st.counter, .acquire)); + out.clearRetainingCapacity(); + try d.pausedStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + try d.resumeThread(tid); + sleepMs(5); + try testing.expect(@atomicLoad(u32, &st.counter, .acquire) != frozen); + @atomicStore(bool, &st.stop, true, .release); + th.join(); +} + +const LockState = struct { + tid: std.atomic.Value(u32) = .init(0), + release: std.atomic.Value(bool) = .init(false), + unlocked: std.atomic.Value(bool) = .init(false), + stop: std.atomic.Value(bool) = .init(false), + io: std.Io, +}; + +fn lockHolderMain(st: *LockState) void { + const di = std.debug.getSelfDebugInfo() catch return; + di.rwlock.lockUncancelable(st.io); + st.tid.store(selfTid(), .release); + while (!st.release.load(.acquire)) sleepMs(1); + di.rwlock.unlock(st.io); + st.unlocked.store(true, .release); + while (!st.stop.load(.acquire)) sleepMs(1); +} + +test "a target parked while holding std.debug's lock is Busy, not a deadlock" { + if (comptime !@hasField(std.debug.SelfInfo, "rwlock")) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var st: LockState = .{ .io = testing.io }; + const th = try std.Thread.spawn(.{}, lockHolderMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + try testing.expect(tries < 2000); + sleepMs(1); + } + const tid = st.tid.load(.acquire); + // No allocation while the holder has the lock: `testing.allocator` + // records a stack trace per allocation, which needs that same lock. + var buf: [16 * 1024]u8 = undefined; + var w: Writer = .fixed(&buf); + try testing.expectError(error.Busy, d.threadStack(tid, &w)); + try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); + // Registers need no unwind and are still available. + try d.threadRegs(tid, &w); + try testing.expect(std.mem.indexOf(u8, w.buffered(), "pc 0x") != null); + // Handshake, not a sleep: a slow holder would otherwise still hold the + // lock and the next capture would legitimately be Busy again. + st.release.store(true, .release); + tries = 0; + while (!st.unlocked.load(.acquire)) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + w = .fixed(&buf); + try d.threadStack(tid, &w); + try testing.expect(std.mem.indexOf(u8, w.buffered(), "lockHolderMain") != null); + st.stop.store(true, .release); + th.join(); +} + +const TrapState = struct { + tid: std.atomic.Value(u32) = .init(0), + counter: u32 = 0, +}; + +noinline fn trapThreadMain(st: *TrapState) void { + st.tid.store(selfTid(), .release); + @breakpoint(); + _ = @atomicRmw(u32, &st.counter, .Add, 1, .acq_rel); +} + +test "breakpoint: pause, inspect, resume" { + if (arch != .x86_64 and !arch.isAARCH64()) return error.SkipZigTest; + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + try d.enableBreakpoints(); + + var st: TrapState = .{}; + const th = try std.Thread.spawn(.{}, trapThreadMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0 or !d.isPaused(st.tid.load(.acquire))) : (tries += 1) { + try testing.expect(tries < 5000); + sleepMs(1); + } + const tid = st.tid.load(.acquire); + try testing.expectEqual(@as(?u32, tid), d.pausedAt(0)); + try testing.expectEqual(@as(?u32, null), d.pausedAt(1)); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.pausedStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "trapThreadMain") != null); + out.clearRetainingCapacity(); + try d.pausedRegs(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "pc 0x") != null); + + // A paused thread can also be captured through the signal path. + out.clearRetainingCapacity(); + try d.threadStack(tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); + + sleepMs(5); + try testing.expectEqual(@as(u32, 0), @atomicLoad(u32, &st.counter, .acquire)); + try d.resumeThread(tid); + th.join(); + try testing.expectEqual(@as(u32, 1), @atomicLoad(u32, &st.counter, .acquire)); + try testing.expect(!d.isPaused(tid)); + try testing.expectEqual(@as(?u32, null), d.pausedAt(0)); + try testing.expectError(error.NotPaused, d.resumeThread(tid)); + try testing.expectError(error.NotPaused, d.pausedStack(tid, &out.writer)); + d.disableBreakpoints(); +} + +/// Stands in for `FullPanic`'s call: the first trace address is the return +/// address into the panicking function. +noinline fn panicLike(msg: []const u8) bool { + return recordPanic(msg, @returnAddress()); +} + +test "panic record path" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + try d.init(testOptions(&text_buf)); + defer d.deinit(); + defer resetPanicRecord(); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + try d.panicMessage(&out.writer); + try testing.expectEqualStrings("", out.written()); + try testing.expect(!d.panicHeld()); + try testing.expectError(error.NoPanic, d.panicContinue()); + + try testing.expect(panicLike("something broke")); + try testing.expect(!recordPanic("nested", null)); + try testing.expectEqual(selfTid(), panic_tid); + + try d.panicMessage(&out.writer); + try testing.expectEqualStrings("something broke", out.written()); + out.clearRetainingCapacity(); + try d.panicStack(&out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "#0 0x") != null); + try testing.expect(std.mem.indexOf(u8, out.written(), "test.panic record path") != null); + try testing.expect(!d.panicHeld()); + try testing.expectError(error.NoPanic, d.panicContinue()); + + // A long message is truncated, not overflowed. + resetPanicRecord(); + const long = [_]u8{'x'} ** (panic_msg_cap + 100); + try testing.expect(recordPanic(&long, null)); + out.clearRetainingCapacity(); + try d.panicMessage(&out.writer); + try testing.expectEqual(@as(usize, panic_msg_cap), out.written().len); +} + +const MaskState = struct { + tid: std.atomic.Value(u32) = .init(0), + unblock: std.atomic.Value(bool) = .init(false), + stop: std.atomic.Value(bool) = .init(false), + signal: linux.SIG, +}; + +fn maskedThreadMain(st: *MaskState) void { + var set = linux.sigemptyset(); + linux.sigaddset(&set, st.signal); + _ = linux.sigprocmask(linux.SIG.BLOCK, &set, null); + st.tid.store(selfTid(), .release); + while (!st.unblock.load(.acquire)) sleepMs(1); + _ = linux.sigprocmask(linux.SIG.UNBLOCK, &set, null); + while (!st.stop.load(.acquire)) sleepMs(1); +} + +test "capture timeout on a thread with the signal masked" { + var text_buf: [16 * 1024]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.capture_timeout_ns = 50 * std.time.ns_per_ms; + try d.init(opts); + defer d.deinit(); + + var st: MaskState = .{ .signal = d.capture_signal }; + const th = try std.Thread.spawn(.{}, maskedThreadMain, .{&st}); + var tries: usize = 0; + while (st.tid.load(.acquire) == 0) : (tries += 1) { + try testing.expect(tries < 2000); + sleepMs(1); + } + const masked_tid = st.tid.load(.acquire); + + var out: Writer.Allocating = .init(testing.allocator); + defer out.deinit(); + const t0 = monotonicNs(); + try testing.expectError(error.Timeout, d.threadStack(masked_tid, &out.writer)); + try testing.expect(monotonicNs() - t0 >= 50 * std.time.ns_per_ms); + try testing.expectEqual(cap_idle, d.capture.state.load(.acquire)); + + // The process is healthy: another thread can still be captured... + var spin: SpinState = .{}; + const spinner = try std.Thread.spawn(.{}, spinThreadMain, .{&spin}); + const spin_tid = waitForTid(&spin); + try testing.expect(spin_tid != 0); + out.clearRetainingCapacity(); + try d.threadStack(spin_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + + // ...and the late delivery of the pending signal is harmless. + st.unblock.store(true, .release); + sleepMs(20); + out.clearRetainingCapacity(); + try d.threadStack(spin_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "spinHere") != null); + out.clearRetainingCapacity(); + try d.threadStack(masked_tid, &out.writer); + try testing.expect(std.mem.indexOf(u8, out.written(), "maskedThreadMain") != null); + + @atomicStore(bool, &spin.stop, true, .release); + spinner.join(); + st.stop.store(true, .release); + th.join(); +} + +test "options validation and single instance" { + var text_buf: [4096]u8 = undefined; + var d: Debug = undefined; + var opts = testOptions(&text_buf); + opts.capture_signal = 5; + try testing.expectError(error.InvalidOptions, d.init(opts)); + opts = testOptions(&text_buf); + opts.max_paused = max_paused_cap + 1; + try testing.expectError(error.InvalidOptions, d.init(opts)); + try d.init(testOptions(&text_buf)); + defer d.deinit(); + var d2: Debug = undefined; + try testing.expectError(error.AlreadyInitialized, d2.init(testOptions(&text_buf))); +} -- cgit v1.3