From 169ba71c6ca54be373a3bbccf17ba199bba17ca8 Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Thu, 19 Feb 2026 23:21:06 -0300 Subject: added option to run with gdb --- build.zig | 53 ++++-- build.zig.zon | 4 + spell.zig | 28 --- src/elfo-pretty.zig | 480 ------------------------------------------------- src/elfo.zig | 500 ++++++++++++++++++++++++++++++++++++++++++++++++++++ tests/test.zig | 8 + 6 files changed, 555 insertions(+), 518 deletions(-) delete mode 100644 spell.zig delete mode 100644 src/elfo-pretty.zig create mode 100644 src/elfo.zig create mode 100644 tests/test.zig diff --git a/build.zig b/build.zig index 10055a2..7b1e650 100644 --- a/build.zig +++ b/build.zig @@ -12,11 +12,12 @@ pub fn build(b: *std.Build) !void { }); const elfo = b.addExecutable(.{ - .name = "elfo-pretty", + .name = "elfo", .root_module = b.createModule(.{ - .root_source_file = b.path("src/elfo-pretty.zig"), + .root_source_file = b.path("src/elfo.zig"), .target = target, .optimize = optimize, + .link_libc = true, }), }); @@ -47,15 +48,45 @@ pub fn build(b: *std.Build) !void { b.installArtifact(elfo); b.installArtifact(gloves); - const run_elfo_cmd = b.addRunArtifact(elfo); - const run_elfo_step = b.step("elfo", "See the pretty elfo!"); - run_elfo_step.dependOn(&run_elfo_cmd.step); + const gdb_opt = b.option(bool, "gdb", "Run executable under gdb") orelse false; + const comp_opts = b.addOptions(); + comp_opts.addOption(bool, "gdb", gdb_opt); - const run_gloves_cmd = b.addRunArtifact(gloves); + const unit_tests = b.addTest(.{ + .root_module = b.createModule(.{ + .optimize = optimize, + .target = target, + .root_source_file = b.path("tests/test.zig"), + }), + }); + + const test_step = b.step("test", "Test the library"); + + const run_elfo_step = b.step("elfo", "See the pretty elfo!"); const run_gloves_step = b.step("gloves", "See the pretty gloves!"); - run_gloves_step.dependOn(&run_gloves_cmd.step); - // TODO: add flag to run on gdb + const run_steps = [_]struct { *std.Build.Step, *std.Build.Step.Compile }{ + .{ test_step, unit_tests }, + .{ run_elfo_step, elfo }, + .{ run_gloves_step, gloves }, + }; + const gdbscript = b.addWriteFile("gdbscript", + \\r + ); + + for (run_steps) |run| { + run.@"1".root_module.addOptions("meta", comp_opts); + if (gdb_opt) { + const run_gdb = b.addSystemCommand(&.{"gdb"}); + run_gdb.addArg("-x"); + run_gdb.addFileArg(gdbscript.getDirectory().path(b, "gdbscript")); + run_gdb.addArtifactArg(run.@"1"); + run.@"0".dependOn(&run_gdb.step); + } else { + const run_unit = b.addRunArtifact(run.@"1"); + run.@"0".dependOn(&run_unit.step); + } + } } fn gum_stalker(b: *std.Build, opts: struct { @@ -87,7 +118,8 @@ fn gum_stalker(b: *std.Build, opts: struct { // C-only module (translate-c can't handle GLib's _Pragma macros) const c_mod = b.createModule(.{ .target = opts.target, - .optimize = opts.optimize, + // .optimize = opts.optimize, + .optimize = .Debug, .link_libc = true, }); @@ -239,7 +271,8 @@ fn capstone(b: *std.Build, opts: struct { const translate_c = b.addTranslateC(.{ .root_source_file = capstone_dep.path("include/capstone/capstone.h"), .target = opts.target, - .optimize = opts.optimize, + // .optimize = opts.optimize, + .optimize = .Debug, .link_libc = true, }); diff --git a/build.zig.zon b/build.zig.zon index b63f625..4906467 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -6,6 +6,10 @@ .url = "git+https://github.com/capstone-engine/capstone?ref=6.0.0-Alpha6#484857da5dc67f7d0e0a01c36b0ebc37a349e0fd", .hash = "N-V-__8AAI-jMgXy8ymEREsD2WbO6CpmH9mIpULsgYu3r56r", }, + .@"frida-gum" = .{ + .url = "git+https://github.com/frida/frida-gum.git#7804914181117f05d20d3b143f7c13935658777f", + .hash = "N-V-__8AAFeo1AEV9-c5BwB6KQi5gLoDP-sj6d2duyERG2u-", + }, }, .paths = .{""}, .fingerprint = 0xdc1ade2e38c84164, diff --git a/spell.zig b/spell.zig deleted file mode 100644 index 5439a6c..0000000 --- a/spell.zig +++ /dev/null @@ -1,28 +0,0 @@ -const std = @import("std"); -const config = @import("config"); -const cs = @import("capstone"); -const elfo = @import("./src/elfo.zig"); - -var gpa: std.heap.GeneralPurposeAllocator(.{}) = .init; -const alloc = gpa.allocator(); - -// FIXME: call capstone on the sections -// FIXME: missing reading the elf sections - -pub fn main() !void { - std.debug.print("abracadabra! -> {s}\n", .{config.vessel}); - // std.debug.print("abracadabra! -> {x} {s}\n", .{ config.vessel_contents, config.vessel }); - - var child = std.process.Child.init(&.{config.vessel}, alloc); - try child.spawn(); - child.stdin_behavior = .Pipe; - child.stdout_behavior = .Pipe; - child.stderr_behavior = .Pipe; - - // var buf: [8]u8 = undefined; - - // const bytes = try child.stdout.?.read(&buf); - // std.debug.print("stdout {x}\n", .{bytes}); - - _ = try child.wait(); -} diff --git a/src/elfo-pretty.zig b/src/elfo-pretty.zig deleted file mode 100644 index 1482670..0000000 --- a/src/elfo-pretty.zig +++ /dev/null @@ -1,480 +0,0 @@ -const std = @import("std"); -const cs = @import("capstone"); - -pub fn main(init: std.process.Init) !void { - var args = try init.minimal.args.iterateAllocator(init.gpa); - defer args.deinit(); - _ = args.next(); // skip argv[0] - - var buffer: [64]u8 = undefined; - const stderr = try init.io.lockStderr(&buffer, .escape_codes); - - // TODO: finish passing custom alloc operations here - // const mem_config: cs.cs_opt_mem = undefined; - // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK); - - var handle: usize = undefined; - defer _ = cs.cs_close(@ptrCast(&handle)); - // TODO: read the arch from the elf so we open the equivalent capstone handle - std.debug.assert(cs.cs_open(cs.CS_ARCH_X86, cs.CS_MODE_64, @ptrCast(&handle)) == cs.CS_ERR_OK); - - try printElf( - init.gpa, - init.io, - args.next() orelse "./study-samples/split", - stderr.terminal(), - handle, - .{ - // .show_unaddressable_sections = true, - // .skip_sections_content = true, - }, - ); -} - -pub fn printElf( - gpa: std.mem.Allocator, - io: std.Io, - path: []const u8, - term: std.Io.Terminal, - handle: usize, - options: struct { - show_unaddressable_sections: bool = false, - skip_sections_content: bool = false, - }, -) !void { - const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); - const bw = term.writer; - defer bw.flush() catch {}; - var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); - defer gpa.free(buffer); - - var reader = f.reader(io, buffer); - const header = try std.elf.Header.read(&reader.interface); - - const shstrtab = blk: { - var section_it = header.iterateSectionHeaders(&reader); - var section_idx: u32 = 0; - while (try section_it.next()) |s| { - defer section_idx += 1; - if (section_idx == header.shstrndx) { - std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); - break :blk s; - } - } - break :blk null; - }; - - const elf_shstrtab_slice = blk: { - if (shstrtab == null) - break :blk null; - - try reader.seekTo(shstrtab.?.sh_offset); - const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size); - break :blk slice; - }; - defer { - if (elf_shstrtab_slice != null) - gpa.free(elf_shstrtab_slice.?); - } - - const strtab = blk: { - if (elf_shstrtab_slice == null) - break :blk null; - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql( - u8, - ".strtab", - std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0), - )) - // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0) - break :blk s; - } - break :blk null; - }; - - const elf_strtab_slice = blk: { - if (strtab == null) - break :blk null; - - try reader.seekTo(strtab.?.sh_offset); - const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size); - break :blk slice; - }; - - defer { - if (elf_strtab_slice != null) - gpa.free(elf_strtab_slice.?); - } - - var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8); - { - if (elf_shstrtab_slice != null) { - var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0); - while (str_it.next()) |str| { - const owned_str = try gpa.alloc(u8, str.len); - @memcpy(owned_str, str); - try strs.append(gpa, owned_str); - } - } - } - defer { - for (strs.items) |s| { - gpa.free(s); - } - strs.deinit(gpa); - } - - var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8); - { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |section| { - try sections.append(gpa, section); - } - std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct { - pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { - return x.sh_addr < y.sh_addr; - } - }.inner); - } - defer sections.deinit(gpa); - - const symtab = blk: { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_SYMTAB) { - try bw.print("sym: {any}\n", .{s}); - break :blk s; - } - } - break :blk null; - }; - - const dynsym = blk: { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_DYNSYM) { - try bw.print("sym: {any}\n", .{s}); - break :blk s; - } - } - break :blk null; - }; - try bw.print("dynsym: {any}\n", .{dynsym}); - - var symbols_index = blk: { - var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); - if (symtab != null) { - var sym_it = iterateSymbols(header, &reader, symtab.?); - while (try sym_it.next()) |s| { - const t = s.st_info & 0xf; - const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); - const owned_name = try gpa.alloc(u8, name.len); - @memcpy(owned_name, name); - try syms.append(gpa, .{ - .start = s.st_value, - .end = s.st_value + s.st_size, - .name = owned_name, - .kind = t, - }); - } - } - - // the check on elf_strtab_slice might not be necessary - if (dynsym != null and elf_strtab_slice != null) { - var sym_it = iterateSymbols(header, &reader, dynsym.?); - while (try sym_it.next()) |s| { - const t = s.st_info & 0xf; - const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); - const owned_name = try gpa.alloc(u8, name.len); - @memcpy(owned_name, name); - try syms.append(gpa, .{ - .start = s.st_value, - .end = s.st_value + s.st_size, - .name = owned_name, - .kind = t, - }); - } - } - std.mem.sort(SymbolRange, syms.items, {}, struct { - fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { - return x.start < y.start; - } - }.inner); - break :blk syms; - }; - defer { - for (symbols_index.items) |sym| { - gpa.free(sym.name); - } - symbols_index.deinit(gpa); - } - - for (symbols_index.items) |sym| { - if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) - try bw.print("{s} {x}-{x}\n", .{ sym.name, sym.start, sym.end }); - } - - for (sections.items) |section| { - if (section.sh_size > 0 and section.sh_addr > 0) { - try term.setColor(.reset); - try term.setColor(.dim); - try bw.print("\n{x}-{x} (t: {x}) -- ", .{ - section.sh_addr, - section.sh_addr + section.sh_size, - section.sh_type, - }); - try term.setColor(.bright_green); - if (elf_shstrtab_slice != null) - try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); - try bw.print("\n", .{}); - try term.setColor(.reset); - - // -- - try reader.seekTo(section.sh_offset); - - if (buffer.len < section.sh_size) { - buffer = try gpa.realloc(buffer, section.sh_size); - reader = f.reader(io, buffer); - } - const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { - switch (e) { - error.EndOfStream => { - try bw.print("failed\n", .{}); - break :blk null; - }, - error.ReadFailed => unreachable, - } - }; - // TODO: this heuristic is probably wrong - if (section_slice != null and !options.skip_sections_content) { - if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) { - const instrs: []cs.cs_insn = blk: { - var insn: [*]cs.cs_insn = undefined; - // TODO: use iter API - // https://www.capstone-engine.org/iteration.html - // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn)); - const count = cs.cs_disasm(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn)); - break :blk insn[0..count]; - }; - - try dumpInstr(gpa, bw, term, instrs, symbols_index.items); - } else { - try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); - } - } - } - } - - if (options.show_unaddressable_sections) { - for (sections.items) |section| { - if (section.sh_size > 0 and section.sh_addr == 0) { - try term.setColor(.reset); - try term.setColor(.dim); - try bw.print("{x}-{x} (t: {x}) -- ", .{ - section.sh_addr, - section.sh_addr + section.sh_size, - section.sh_type, - }); - try term.setColor(.bright_cyan); - if (elf_shstrtab_slice != null) - try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); - try bw.print("\n", .{}); - try term.setColor(.reset); - // -- - - try reader.seekTo(section.sh_offset); - - if (buffer.len < section.sh_size) { - buffer = try gpa.realloc(buffer, section.sh_size); - reader = f.reader(io, buffer); - } - const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { - switch (e) { - error.EndOfStream => { - break :blk null; - }, - error.ReadFailed => unreachable, - } - }; - if (section_slice != null and !options.skip_sections_content) { - try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); - } - } - } - } -} - -fn allocComment( - gpa: std.mem.Allocator, - code: []u8, - symbols: []SymbolRange, -) !?[]u8 { - var iter = std.mem.splitAny(u8, code, " \t[],+-"); - while (iter.next()) |s| { - if (std.mem.startsWith(u8, s, "0x")) { - // todo split at the zero char at the end of string - const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { - std.debug.print("{any}\n", .{e}); - std.debug.dumpHex(s); - break :blk 0; - }; - if (v > 0) { - const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { - fn inner(a: u64, sym: SymbolRange) std.math.Order { - return std.math.order(a, sym.start); - } - }.inner); - if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { - // if (idx > 0) - // idx -= 1; - const d = v - symbols[idx].start; - if (d > 0) - return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); - return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); - } - } - } - } - - return null; -} - -fn dumpInstr( - gpa: std.mem.Allocator, - bw: *std.Io.Writer, - term: std.Io.Terminal, - instrs: []cs.cs_insn, - symbols: []SymbolRange, -) !void { - for (instrs) |instr| { - const addr = instr.address; - const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct { - fn inner(a: u64, sym: SymbolRange) std.math.Order { - return std.math.order(a, sym.start); - } - }.inner); - - if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { - try term.setColor(.blue); - try bw.print("\n{x:0>16} {s}:\n", .{ - addr, - symbols[idx].name, - }); - try term.setColor(.reset); - } - try term.setColor(.dim); - try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); - try term.setColor(.reset); - // if(instr.detail.) - try term.setColor(.bright_green); - try bw.print("{s} ", .{instr.mnemonic}); - try term.setColor(.reset); - try bw.print("{s}", .{instr.op_str}); - const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols); - defer { - if (asm_comment != null) - gpa.free(asm_comment.?); - } - if (asm_comment != null and asm_comment.?.len > 0) { - try term.setColor(.blue); - try bw.print(" <{s}>", .{asm_comment.?}); - } - try bw.print("\n", .{}); - try term.setColor(.reset); - } -} - -/// Prints a hexadecimal view of the bytes, returning any error that occurs. -pub fn dumpHexFallible( - _: type, - bw: *std.Io.Writer, - term: std.Io.Terminal, - bytes: []const u8, - offset: u64, -) !void { - // @breakpoint(); - const nbytes = 16; - var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); - while (chunks.next()) |window| { - // 1. Print the address. - const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; - try term.setColor(.dim); - // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more. - // Also, make sure all lines are aligned by padding the address. - try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); - try term.setColor(.reset); - - // 2. Print the bytes. - for (window, 0..) |byte, index| { - try bw.print("{X:0>2} ", .{byte}); - if (index == 7) try bw.writeByte(' '); - } - try bw.writeByte(' '); - if (window.len < 16) { - var missing_columns = (16 - window.len) * 3; - if (window.len < 8) missing_columns += 1; - try bw.splatByteAll(' ', missing_columns); - } - - const window_bytes: []const u8 = @ptrCast(@alignCast(window)); - - // 3. Print the characters. - for (window_bytes) |byte| { - if (std.ascii.isPrint(byte)) { - try bw.writeByte(byte); - } else { - - // Let's print some common control codes as graphical Unicode symbols. - // We don't want to do this for all control codes because most control codes apart from - // the ones that Zig has escape sequences for are likely not very useful to print as symbols. - switch (byte) { - '\n' => try bw.writeAll("␊"), - '\r' => try bw.writeAll("␍"), - '\t' => try bw.writeAll("␉"), - else => try bw.writeByte('.'), - } - } - } - try bw.writeByte('\n'); - } -} - -const SymbolRange = struct { - start: u64, - end: u64, - name: []u8, - kind: u8, -}; - -fn iterateSymbols( - h: std.elf.Header, - file_reader: *std.Io.File.Reader, - symtab: std.elf.Elf64_Shdr, -) SymbolIterator { - return .{ - .elf_header = h, - .file_reader = file_reader, - .symtab = symtab, - }; -} - -const SymbolIterator = struct { - elf_header: std.elf.Header, - file_reader: *std.Io.File.Reader, - symtab: std.elf.Elf64_Shdr, - index: usize = 0, - - pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { - defer it.index += 1; - - const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); - const offset = it.symtab.sh_offset + size * it.index; - - if (offset >= (it.symtab.sh_size + it.symtab.sh_offset)) - return null; - - try it.file_reader.seekTo(offset); - return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian); - } -}; diff --git a/src/elfo.zig b/src/elfo.zig new file mode 100644 index 0000000..20f798a --- /dev/null +++ b/src/elfo.zig @@ -0,0 +1,500 @@ +const std = @import("std"); +const cs = @import("capstone"); +const meta_opts = @import("meta"); + +pub fn main(init: std.process.Init) !void { + var args = try init.minimal.args.iterateAllocator(init.gpa); + defer args.deinit(); + _ = args.next(); // skip argv[0] + + var buffer: [64]u8 = undefined; + const stderr = try init.io.lockStderr(&buffer, .escape_codes); + + // TODO: finish passing custom alloc operations here + // const mem_config: cs.cs_opt_mem = undefined; + // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK); + + if (meta_opts.gdb) @breakpoint(); + + try printElf( + init.gpa, + init.io, + args.next() orelse "./study-samples/split", + stderr.terminal(), + .{ + .show_unaddressable_sections = true, + // .skip_sections_content = true, + }, + ); +} + +pub fn printElf( + gpa: std.mem.Allocator, + io: std.Io, + path: []const u8, + term: std.Io.Terminal, + options: struct { + show_unaddressable_sections: bool = false, + skip_sections_content: bool = false, + }, +) !void { + const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); + const bw = term.writer; + defer bw.flush() catch {}; + var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); + defer gpa.free(buffer); + + var reader = f.reader(io, buffer); + const header = try std.elf.Header.read(&reader.interface); + + var handle: usize = undefined; + defer _ = cs.cs_close(@ptrCast(&handle)); + const opts: struct { arch: u64, mode: u64 } = switch (header.machine) { + .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 }, + // The arm mode isn't working with symbols on the .text section correctly + .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = cs.CS_MODE_ARM }, + else => { + std.debug.print("found machine: {any}\n", .{header.machine}); + @panic("unhandled arch"); + }, + }; + + std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK); + + const shstrtab = blk: { + var section_it = header.iterateSectionHeaders(&reader); + var section_idx: u32 = 0; + while (try section_it.next()) |s| { + defer section_idx += 1; + if (section_idx == header.shstrndx) { + std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); + break :blk s; + } + } + break :blk null; + }; + + const elf_shstrtab_slice = blk: { + if (shstrtab == null) + break :blk null; + + try reader.seekTo(shstrtab.?.sh_offset); + const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size); + break :blk slice; + }; + defer { + if (elf_shstrtab_slice != null) + gpa.free(elf_shstrtab_slice.?); + } + + const strtab = blk: { + if (elf_shstrtab_slice == null) + break :blk null; + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |s| { + if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql( + u8, + ".strtab", + std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0), + )) + // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0) + break :blk s; + } + break :blk null; + }; + + const elf_strtab_slice = blk: { + if (strtab == null) + break :blk null; + + try reader.seekTo(strtab.?.sh_offset); + const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size); + break :blk slice; + }; + + defer { + if (elf_strtab_slice != null) + gpa.free(elf_strtab_slice.?); + } + + var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8); + { + if (elf_shstrtab_slice != null) { + var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0); + while (str_it.next()) |str| { + const owned_str = try gpa.alloc(u8, str.len); + @memcpy(owned_str, str); + try strs.append(gpa, owned_str); + } + } + } + defer { + for (strs.items) |s| { + gpa.free(s); + } + strs.deinit(gpa); + } + + var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8); + { + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |section| { + try sections.append(gpa, section); + } + std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct { + pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { + return x.sh_addr < y.sh_addr; + } + }.inner); + } + defer sections.deinit(gpa); + + const symtab = blk: { + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |s| { + if (s.sh_type == std.elf.SHT_SYMTAB) { + + // x64: symtab: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4256, .sh_size = 1680, .sh_link = 27, .sh_info = 45, .sh_addralign = 8, .sh_entsize = 24 } + // arm32: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4264, .sh_size = 1856, .sh_link = 27, .sh_info = 87, .sh_addralign = 4, .sh_entsize = 16 } + try bw.print("symtab: {any}\n", .{s}); + break :blk s; + } + } + break :blk null; + }; + + const dynsym = blk: { + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |s| { + if (s.sh_type == std.elf.SHT_DYNSYM) { + // try bw.print("sym: {any}\n", .{s}); + break :blk s; + } + } + break :blk null; + }; + // try bw.print("dynsym: {any}\n", .{dynsym}); + + var symbols_index = blk: { + var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); + if (symtab != null) { + var sym_it = iterateSymbols(header, &reader, symtab.?); + while (try sym_it.next()) |s| { + const t = s.st_info & 0xf; + const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); + const owned_name = try gpa.alloc(u8, name.len); + @memcpy(owned_name, name); + try syms.append(gpa, .{ + .start = s.st_value, + .end = s.st_value + s.st_size, + .name = owned_name, + .kind = t, + }); + } + } + + // the check on elf_strtab_slice might not be necessary + if (dynsym != null and elf_strtab_slice != null) { + var sym_it = iterateSymbols(header, &reader, dynsym.?); + while (try sym_it.next()) |s| { + const t = s.st_info & 0xf; + const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); + const owned_name = try gpa.alloc(u8, name.len); + @memcpy(owned_name, name); + try syms.append(gpa, .{ + .start = s.st_value, + .end = s.st_value + s.st_size, + .name = owned_name, + .kind = t, + }); + } + } + std.mem.sort(SymbolRange, syms.items, {}, struct { + fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { + return x.start < y.start; + } + }.inner); + break :blk syms; + }; + defer { + for (symbols_index.items) |sym| { + gpa.free(sym.name); + } + symbols_index.deinit(gpa); + } + + for (symbols_index.items) |sym| { + if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) + try bw.print("{s} {x}-{x}\n", .{ sym.name, sym.start, sym.end }); + } + + for (sections.items) |section| { + if (section.sh_size > 0 and section.sh_addr > 0) { + try term.setColor(.reset); + try term.setColor(.dim); + try bw.print("\n{x}-{x} (t: {x}) -- ", .{ + section.sh_addr, + section.sh_addr + section.sh_size, + section.sh_type, + }); + try term.setColor(.bright_green); + if (elf_shstrtab_slice != null) + try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); + try bw.print("\n", .{}); + try term.setColor(.reset); + + // -- + try reader.seekTo(section.sh_offset); + + if (buffer.len < section.sh_size) { + buffer = try gpa.realloc(buffer, section.sh_size); + reader = f.reader(io, buffer); + } + const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { + switch (e) { + error.EndOfStream => { + try bw.print("failed\n", .{}); + break :blk null; + }, + error.ReadFailed => unreachable, + } + }; + // TODO: this heuristic is probably wrong + if (section_slice != null and !options.skip_sections_content) { + if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) { + const instrs: []cs.cs_insn = blk: { + var insn: [*]cs.cs_insn = undefined; + // TODO: use iter API + // https://www.capstone-engine.org/iteration.html + // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn)); + const count = cs.cs_disasm(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn)); + break :blk insn[0..count]; + }; + + try dumpInstr(gpa, bw, term, instrs, symbols_index.items); + } else { + try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); + } + } + } + } + + if (options.show_unaddressable_sections) { + for (sections.items) |section| { + if (section.sh_size > 0 and section.sh_addr == 0) { + try term.setColor(.reset); + try term.setColor(.dim); + try bw.print("{x}-{x} (t: {x}) -- ", .{ + section.sh_addr, + section.sh_addr + section.sh_size, + section.sh_type, + }); + try term.setColor(.bright_cyan); + if (elf_shstrtab_slice != null) + try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); + try bw.print("\n", .{}); + try term.setColor(.reset); + // -- + + try reader.seekTo(section.sh_offset); + + if (buffer.len < section.sh_size) { + buffer = try gpa.realloc(buffer, section.sh_size); + reader = f.reader(io, buffer); + } + const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { + switch (e) { + error.EndOfStream => { + break :blk null; + }, + error.ReadFailed => unreachable, + } + }; + if (section_slice != null and !options.skip_sections_content) { + try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); + } + } + } + } +} + +fn allocComment( + gpa: std.mem.Allocator, + code: []u8, + symbols: []SymbolRange, +) !?[]u8 { + var iter = std.mem.splitAny(u8, code, " \t[],+-"); + while (iter.next()) |s| { + if (std.mem.startsWith(u8, s, "0x")) { + // todo split at the zero char at the end of string + const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { + std.debug.print("{any}\n", .{e}); + std.debug.dumpHex(s); + break :blk 0; + }; + // FIXME: this algorithm isn't working to find addresses "inside" symbols + if (v > 0) { + const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { + fn inner(a: u64, sym: SymbolRange) std.math.Order { + return std.math.order(a, sym.start); + } + }.inner); + if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { + // if (idx > 0) + // idx -= 1; + const d = v - symbols[idx].start; + if (d > 0) + return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); + return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); + } + } + } + } + + return null; +} + +fn dumpInstr( + gpa: std.mem.Allocator, + bw: *std.Io.Writer, + term: std.Io.Terminal, + instrs: []cs.cs_insn, + symbols: []SymbolRange, +) !void { + for (instrs) |instr| { + const addr = instr.address; + const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct { + fn inner(a: u64, sym: SymbolRange) std.math.Order { + return std.math.order(a, sym.start); + } + }.inner); + + if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { + try term.setColor(.blue); + try bw.print("\n{x:0>16} {s}:\n", .{ + addr, + symbols[idx].name, + }); + try term.setColor(.reset); + } + try term.setColor(.dim); + try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); + try term.setColor(.reset); + // if(instr.detail.) + try term.setColor(.bright_green); + try bw.print("{s} ", .{instr.mnemonic}); + try term.setColor(.reset); + + const mnemonic_strlen: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?); + const mnemonic_pad: u64 = 10; + for (0..(mnemonic_pad - mnemonic_strlen)) |_| { + try bw.printAsciiChar(' ', .{}); + } + try bw.print("{s}", .{instr.op_str}); + const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols); + defer { + if (asm_comment != null) + gpa.free(asm_comment.?); + } + if (asm_comment != null and asm_comment.?.len > 0) { + try term.setColor(.blue); + try bw.print(" <{s}>", .{asm_comment.?}); + } + try bw.print("\n", .{}); + try term.setColor(.reset); + } +} + +/// Prints a hexadecimal view of the bytes, returning any error that occurs. +pub fn dumpHexFallible( + _: type, + bw: *std.Io.Writer, + term: std.Io.Terminal, + bytes: []const u8, + offset: u64, +) !void { + // @breakpoint(); + const nbytes = 16; + var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); + while (chunks.next()) |window| { + // 1. Print the address. + const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; + try term.setColor(.dim); + // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more. + // Also, make sure all lines are aligned by padding the address. + try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); + try term.setColor(.reset); + + // 2. Print the bytes. + for (window, 0..) |byte, index| { + try bw.print("{X:0>2} ", .{byte}); + if (index == 7) try bw.writeByte(' '); + } + try bw.writeByte(' '); + if (window.len < 16) { + var missing_columns = (16 - window.len) * 3; + if (window.len < 8) missing_columns += 1; + try bw.splatByteAll(' ', missing_columns); + } + + const window_bytes: []const u8 = @ptrCast(@alignCast(window)); + + // 3. Print the characters. + for (window_bytes) |byte| { + if (std.ascii.isPrint(byte)) { + try bw.writeByte(byte); + } else { + + // Let's print some common control codes as graphical Unicode symbols. + // We don't want to do this for all control codes because most control codes apart from + // the ones that Zig has escape sequences for are likely not very useful to print as symbols. + switch (byte) { + '\n' => try bw.writeAll("␊"), + '\r' => try bw.writeAll("␍"), + '\t' => try bw.writeAll("␉"), + else => try bw.writeByte('.'), + } + } + } + try bw.writeByte('\n'); + } +} + +const SymbolRange = struct { + start: u64, + end: u64, + name: []u8, + kind: u8, +}; + +fn iterateSymbols( + h: std.elf.Header, + file_reader: *std.Io.File.Reader, + symtab: std.elf.Elf64_Shdr, +) SymbolIterator { + return .{ + .elf_header = h, + .file_reader = file_reader, + .symtab = symtab, + }; +} + +const SymbolIterator = struct { + elf_header: std.elf.Header, + file_reader: *std.Io.File.Reader, + symtab: std.elf.Elf64_Shdr, + index: usize = 0, + + pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { + defer it.index += 1; + + const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); + const offset = it.symtab.sh_offset + size * it.index; + + if (offset >= (it.symtab.sh_size + it.symtab.sh_offset)) + return null; + + try it.file_reader.seekTo(offset); + return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian); + } +}; diff --git a/tests/test.zig b/tests/test.zig new file mode 100644 index 0000000..8351b50 --- /dev/null +++ b/tests/test.zig @@ -0,0 +1,8 @@ +const std = @import("std"); +const opts = @import("meta"); +const assert = std.debug.assert; + +test "always pass" { + if (opts.gdb) @breakpoint(); + assert(true); +} -- cgit v1.3