diff options
| author | Gabriel Schneider <[email protected]> | 2026-04-06 17:15:29 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-06-08 23:05:25 -0300 |
| commit | d578c82c7fb3a2ac9557926ae6690b4f13534c22 (patch) | |
| tree | e17774c89dc96e32ecdb97dda3630881a4b96a9d /src/elfo.zig | |
| parent | f014ba6efb32a80d4f4790880fa88b907b82d756 (diff) | |
| download | codenomicon-d578c82c7fb3a2ac9557926ae6690b4f13534c22.tar.gz codenomicon-d578c82c7fb3a2ac9557926ae6690b4f13534c22.zip | |
a lot of refactoring ig
Diffstat (limited to 'src/elfo.zig')
| -rw-r--r-- | src/elfo.zig | 721 |
1 files changed, 398 insertions, 323 deletions
diff --git a/src/elfo.zig b/src/elfo.zig index e07620c..73b666e 100644 --- a/src/elfo.zig +++ b/src/elfo.zig @@ -1,7 +1,5 @@ const std = @import("std"); const cs = @import("capstone"); -const hooks = @import("hooks.zig"); -const meta = @import("meta"); pub fn main(init: std.process.Init.Minimal) !void { var debug_alloc: std.heap.DebugAllocator(.{}) = .init; @@ -18,36 +16,33 @@ pub fn main(init: std.process.Init.Minimal) !void { defer args.deinit(); _ = args.next(); // skip argv[0] + var path: []const u8 = "./study-samples/split"; + var symbol_filter: ?[]const u8 = null; + var compare = false; + while (args.next()) |arg| { + if (std.mem.eql(u8, arg, "-s")) { + symbol_filter = args.next(); + } else if (std.mem.eql(u8, arg, "--compare")) { + compare = true; + } else { + path = arg; + } + } + var buffer: [64]u8 = undefined; const stderr = try io.lockStderr(&buffer, null); - // const hook = hooks.init( - // if (meta.gdb) .breakpoint else .none, - // // TODO: this needs to be comptime, it'd be cool to don't need that - // std.heap.page_allocator, - // false, - // ); - - // const mem_config: cs.cs_opt_mem = .{ - // .malloc = @ptrCast(&hook.malloc), - // .free = @ptrCast(&hook.free), - // .calloc = @ptrCast(&hook.calloc), - // .realloc = @ptrCast(&hook.realloc), - // // TODO: this function needs to be properly implemented. - // .vsnprintf = @ptrCast(&hook.vsnprintf), - // }; - // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK); - - try printElf( - alloc, - io, - args.next() orelse "./study-samples/split", - stderr.terminal(), - .{ - // .show_unaddressable_sections = true, - // .skip_sections_content = true, - }, - ); + if (compare) { + try compareWithObjdump(alloc, io, path, stderr.terminal()); + } else { + try printElf( + alloc, + io, + path, + stderr.terminal(), + .{ .symbol_filter = symbol_filter }, + ); + } } pub fn printElf( @@ -58,200 +53,85 @@ pub fn printElf( options: struct { show_unaddressable_sections: bool = false, skip_sections_content: bool = false, + symbol_filter: ?[]const u8 = null, }, ) !void { const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); const bw = term.writer; defer bw.flush() catch {}; + var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); defer gpa.free(buffer); - var reader = f.reader(io, buffer); + const header = try std.elf.Header.read(&reader.interface); - var cs_handle: usize = undefined; + var cs_handle = initCapstone(header); defer _ = cs.cs_close(@ptrCast(&cs_handle)); - const cs_opts: struct { arch: u64, mode: u64 } = switch (header.machine) { - .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 }, - // The arm mode isn't working with symbols on the .text section correctly - .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = if (header.is_64) cs.CS_MODE_64 else cs.CS_MODE_32 }, - else => { - std.debug.print("found machine: {any}\n", .{header.machine}); - @panic("unhandled arch"); - }, - }; - - std.debug.assert(cs.cs_open(@intCast(cs_opts.arch), @intCast(cs_opts.mode), @ptrCast(&cs_handle)) == cs.CS_ERR_OK); - const shstrtab = blk: { - var section_it = header.iterateSectionHeaders(&reader); - var section_idx: u32 = 0; - while (try section_it.next()) |s| { - defer section_idx += 1; - if (section_idx == header.shstrndx) { - std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); - break :blk s; - } - } - break :blk null; - }; - - const elf_shstrtab_slice = blk: { - if (shstrtab == null) - break :blk null; - - try reader.seekTo(shstrtab.?.sh_offset); - const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size); - break :blk slice; - }; - defer { - if (elf_shstrtab_slice != null) - gpa.free(elf_shstrtab_slice.?); - } + // Single-pass: collect all section headers and find key sections by type/index + var sections = try collectSections(header, &reader, gpa); + defer sections.deinit(gpa); - const strtab = blk: { - if (elf_shstrtab_slice == null) - break :blk null; - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql( - u8, - ".strtab", - std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0), - )) - // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0) - break :blk s; - } - break :blk null; - }; + // Load section header string table + const shstrtab_data: ?[]u8 = if (sections.shstrtab) |s| blk: { + try reader.seekTo(s.sh_offset); + break :blk try reader.interface.readAlloc(gpa, s.sh_size); + } else null; + defer if (shstrtab_data) |d| gpa.free(d); - const elf_strtab_slice = blk: { - if (strtab == null) - break :blk null; + // Find .strtab by name (requires shstrtab_data) + const strtab = if (shstrtab_data) |data| + findStrtab(sections.all.items, data) + else + null; - try reader.seekTo(strtab.?.sh_offset); - const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size); - break :blk slice; - }; + // Load symbol string table + const strtab_data: ?[]u8 = if (strtab) |s| blk: { + try reader.seekTo(s.sh_offset); + break :blk try reader.interface.readAlloc(gpa, s.sh_size); + } else null; + defer if (strtab_data) |d| gpa.free(d); + // Collect symbols from symtab and dynsym + var symbols = try collectSymbols(gpa, header, &reader, strtab_data, sections); defer { - if (elf_strtab_slice != null) - gpa.free(elf_strtab_slice.?); + for (symbols.items) |sym| gpa.free(sym.name); + symbols.deinit(gpa); } - var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8); - { - if (elf_shstrtab_slice != null) { - var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0); - while (str_it.next()) |str| { - const owned_str = try gpa.alloc(u8, str.len); - @memcpy(owned_str, str); - try strs.append(gpa, owned_str); - } - } - } - defer { - for (strs.items) |s| { - gpa.free(s); + // When filtering by symbol, find the target and only render its disassembly + const filter_sym: ?SymbolRange = if (options.symbol_filter) |name| blk: { + for (symbols.items) |sym| { + if (std.mem.eql(u8, sym.name, name)) break :blk sym; } - strs.deinit(gpa); - } + break :blk null; + } else null; - var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8); - { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |section| { - try sections.append(gpa, section); + if (options.symbol_filter == null) { + for (symbols.items) |sym| { + if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) + try bw.print("{x}-{x} {s}\n", .{ sym.start, sym.end, sym.name }); } - std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct { - pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { - return x.sh_addr < y.sh_addr; - } - }.inner); } - defer sections.deinit(gpa); - const symtab = blk: { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_SYMTAB) { + // Render sections + for (sections.all.items) |section| { + if (section.sh_size == 0) continue; + const addressable = section.sh_addr > 0; + if (!addressable and !options.show_unaddressable_sections) continue; - // x64: symtab: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4256, .sh_size = 1680, .sh_link = 27, .sh_info = 45, .sh_addralign = 8, .sh_entsize = 24 } - // arm32: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4264, .sh_size = 1856, .sh_link = 27, .sh_info = 87, .sh_addralign = 4, .sh_entsize = 16 } - // try bw.print("symtab: {any}\n", .{s}); - break :blk s; - } - } - break :blk null; - }; + const is_exec = addressable and section.sh_type == std.elf.SHT_PROGBITS and + (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0; - const dynsym = blk: { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_DYNSYM) { - // try bw.print("sym: {any}\n", .{s}); - break :blk s; - } + // When filtering by symbol, skip sections that don't contain it + if (filter_sym) |fsym| { + if (!is_exec) continue; + const sec_end = section.sh_addr + section.sh_size; + if (fsym.start < section.sh_addr or fsym.start >= sec_end) continue; } - break :blk null; - }; - // try bw.print("dynsym: {any}\n", .{dynsym}); - var symbols_index = blk: { - var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); - if (symtab != null) { - var sym_it = iterateSymbols(header, &reader, symtab.?); - while (try sym_it.next()) |s| { - const t = s.st_info & 0xf; - const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); - const owned_name = try gpa.alloc(u8, name.len); - @memcpy(owned_name, name); - try syms.append(gpa, .{ - .start = s.st_value, - .end = s.st_value + s.st_size, - .name = owned_name, - .kind = t, - }); - } - } - - // the check on elf_strtab_slice might not be necessary - if (dynsym != null and elf_strtab_slice != null) { - var sym_it = iterateSymbols(header, &reader, dynsym.?); - while (try sym_it.next()) |s| { - const t = s.st_info & 0xf; - const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); - const owned_name = try gpa.alloc(u8, name.len); - @memcpy(owned_name, name); - try syms.append(gpa, .{ - .start = s.st_value, - .end = s.st_value + s.st_size, - .name = owned_name, - .kind = t, - }); - } - } - std.mem.sort(SymbolRange, syms.items, {}, struct { - fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { - return x.start < y.start; - } - }.inner); - break :blk syms; - }; - defer { - for (symbols_index.items) |sym| { - gpa.free(sym.name); - } - symbols_index.deinit(gpa); - } - - for (symbols_index.items) |sym| { - if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) - try bw.print("{x}-{x} {s}\n", .{ sym.start, sym.end, sym.name }); - } - - for (sections.items) |section| { - if (section.sh_size > 0 and section.sh_addr > 0) { + if (options.symbol_filter == null) { try term.setColor(.reset); try term.setColor(.dim); try bw.print("\n{x}-{x} (t: {x}) -- ", .{ @@ -259,124 +139,283 @@ pub fn printElf( section.sh_addr + section.sh_size, section.sh_type, }); - - try term.setColor(.bright_green); - if (elf_shstrtab_slice != null) - try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); + try term.setColor(if (addressable) .bright_green else .bright_cyan); + if (shstrtab_data) |data| + try bw.print("{s}", .{std.mem.sliceTo(data[section.sh_name..], 0)}); try bw.print("\n", .{}); try term.setColor(.reset); + } - // -- - try reader.seekTo(section.sh_offset); + try reader.seekTo(section.sh_offset); + if (buffer.len < section.sh_size) { + buffer = try gpa.realloc(buffer, section.sh_size); + reader = f.reader(io, buffer); + } + const section_slice = reader.interface.take(section.sh_size) catch |e| switch (e) { + error.EndOfStream => { + if (addressable) try bw.print("failed\n", .{}); + continue; + }, + error.ReadFailed => unreachable, + }; - if (buffer.len < section.sh_size) { - buffer = try gpa.realloc(buffer, section.sh_size); - reader = f.reader(io, buffer); - } - const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { - switch (e) { - error.EndOfStream => { - try bw.print("failed\n", .{}); - break :blk null; - }, - error.ReadFailed => unreachable, - } - }; - // TODO: this heuristic is probably wrong - if (section_slice != null and !options.skip_sections_content) { - if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) { - const instrs: []cs.cs_insn = blk: { - var insn: [*]cs.cs_insn = undefined; - // TODO: use iter API - // https://www.capstone-engine.org/iteration.html - // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn)); - const count = cs.cs_disasm(cs_handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn)); - break :blk insn[0..count]; - }; + if (options.skip_sections_content) continue; - try printDisassembly(gpa, bw, term, instrs, symbols_index.items); - } else { - try printHexdump(u64, bw, term, section_slice.?, section.sh_addr); + if (is_exec) { + var insn: [*]cs.cs_insn = undefined; + // TODO: use iter API https://www.capstone-engine.org/iteration.html + const count = cs.cs_disasm(cs_handle, section_slice.ptr, section_slice.len, section.sh_addr, 0, @ptrCast(&insn)); + const instrs = insn[0..count]; + + if (filter_sym) |fsym| { + // Find instruction range within the symbol + var start_idx: usize = 0; + var end_idx: usize = instrs.len; + for (instrs, 0..) |instr, i| { + if (instr.address >= fsym.start and start_idx == 0) start_idx = i; + if (instr.address >= fsym.end and fsym.end > fsym.start) { + end_idx = i; + break; + } } + try printDisassembly(gpa, bw, term, instrs[start_idx..end_idx], symbols.items); + } else { + try printDisassembly(gpa, bw, term, instrs, symbols.items); } + } else { + try printHexdump(u64, bw, term, section_slice, section.sh_addr); } } +} - if (options.show_unaddressable_sections) { - for (sections.items) |section| { - if (section.sh_size > 0 and section.sh_addr == 0) { - try term.setColor(.reset); - try term.setColor(.dim); - try bw.print("{x}-{x} (t: {x}) -- ", .{ - section.sh_addr, - section.sh_addr + section.sh_size, - section.sh_type, - }); - try term.setColor(.bright_cyan); - if (elf_shstrtab_slice != null) - try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); - try bw.print("\n", .{}); - try term.setColor(.reset); - // -- +fn compareWithObjdump( + gpa: std.mem.Allocator, + io: std.Io, + path: []const u8, + term: std.Io.Terminal, +) !void { + const bw = term.writer; + defer bw.flush() catch {}; - try reader.seekTo(section.sh_offset); + var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); + defer gpa.free(buffer); - if (buffer.len < section.sh_size) { - buffer = try gpa.realloc(buffer, section.sh_size); - reader = f.reader(io, buffer); - } - const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { - switch (e) { - error.EndOfStream => { - break :blk null; - }, - error.ReadFailed => unreachable, - } - }; - if (section_slice != null and !options.skip_sections_content) { - try printHexdump(u64, bw, term, section_slice.?, section.sh_addr); + const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); + var reader = f.reader(io, buffer); + const header = try std.elf.Header.read(&reader.interface); + + var cs_handle = initCapstone(header); + defer _ = cs.cs_close(@ptrCast(&cs_handle)); + + var sections = try collectSections(header, &reader, gpa); + defer sections.deinit(gpa); + + const shstrtab_data: ?[]u8 = if (sections.shstrtab) |s| blk: { + try reader.seekTo(s.sh_offset); + break :blk try reader.interface.readAlloc(gpa, s.sh_size); + } else null; + defer if (shstrtab_data) |d| gpa.free(d); + + const strtab = if (shstrtab_data) |data| findStrtab(sections.all.items, data) else null; + const strtab_data: ?[]u8 = if (strtab) |s| blk: { + try reader.seekTo(s.sh_offset); + break :blk try reader.interface.readAlloc(gpa, s.sh_size); + } else null; + defer if (strtab_data) |d| gpa.free(d); + + var symbols = try collectSymbols(gpa, header, &reader, strtab_data, sections); + defer { + for (symbols.items) |sym| gpa.free(sym.name); + symbols.deinit(gpa); + } + + // For each function symbol with nonzero size, show elfo vs objdump + for (symbols.items) |sym| { + if (sym.kind != std.elf.STT_FUNC or sym.name.len == 0 or sym.start == sym.end) continue; + + // Find the section containing this symbol + const section = blk: { + for (sections.all.items) |s| { + const is_exec = s.sh_addr > 0 and s.sh_type == std.elf.SHT_PROGBITS and + (s.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0; + if (is_exec and sym.start >= s.sh_addr and sym.start < s.sh_addr + s.sh_size) + break :blk s; + } + continue; + }; + + // Header + try term.setColor(.bright_green); + try bw.writeAll("\n============================================================\n"); + try bw.print(" {s} ({x:0>16} - {x:0>16})\n", .{ sym.name, sym.start, sym.end }); + try bw.writeAll("============================================================\n"); + try term.setColor(.reset); + + // --- elfo output --- + try term.setColor(.blue); + try bw.print("--- elfo ---\n", .{}); + try term.setColor(.reset); + + try reader.seekTo(section.sh_offset); + if (buffer.len < section.sh_size) { + buffer = try gpa.realloc(buffer, section.sh_size); + reader = f.reader(io, buffer); + } + const section_slice = reader.interface.take(section.sh_size) catch |e| switch (e) { + error.EndOfStream => { + try bw.print("failed to read section\n", .{}); + continue; + }, + error.ReadFailed => unreachable, + }; + + { + var insn: [*]cs.cs_insn = undefined; + const count = cs.cs_disasm(cs_handle, section_slice.ptr, section_slice.len, section.sh_addr, 0, @ptrCast(&insn)); + const instrs = insn[0..count]; + + var start_idx: usize = 0; + var end_idx: usize = instrs.len; + for (instrs, 0..) |instr, i| { + if (instr.address >= sym.start and start_idx == 0) start_idx = i; + if (instr.address >= sym.end) { + end_idx = i; + break; } } + try printDisassembly(gpa, bw, term, instrs[start_idx..end_idx], symbols.items); } + bw.flush() catch {}; + + // --- objdump output --- + try term.setColor(.blue); + try bw.print("\n--- objdump ---\n", .{}); + try term.setColor(.reset); + bw.flush() catch {}; + + const start_addr = try std.fmt.allocPrint(gpa, "0x{x}", .{sym.start}); + defer gpa.free(start_addr); + const stop_addr = try std.fmt.allocPrint(gpa, "0x{x}", .{sym.end}); + defer gpa.free(stop_addr); + + const result = std.process.run(gpa, io, .{ + .argv = &.{ + "objdump", "-d", "-M", "intel", "--no-show-raw-insn", + "--start-address", start_addr, + "--stop-address", stop_addr, + path, + }, + }) catch |e| { + try bw.print("failed to run objdump: {any}\n", .{e}); + continue; + }; + defer gpa.free(result.stdout); + defer gpa.free(result.stderr); + + try bw.writeAll(result.stdout); } } -fn allocComment( +// --- Section collection --- + +const SectionInfo = struct { + all: std.ArrayList(std.elf.Elf64_Shdr), + shstrtab: ?std.elf.Elf64_Shdr = null, + symtab: ?std.elf.Elf64_Shdr = null, + dynsym: ?std.elf.Elf64_Shdr = null, + + fn deinit(self: *SectionInfo, gpa: std.mem.Allocator) void { + self.all.deinit(gpa); + } +}; + +fn collectSections( + header: std.elf.Header, + reader: *std.Io.File.Reader, gpa: std.mem.Allocator, - code: []u8, - symbols: []SymbolRange, -) !?[]u8 { - var iter = std.mem.splitAny(u8, code, " \t[],+-"); - while (iter.next()) |s| { - if (std.mem.startsWith(u8, s, "0x")) { - // todo split at the zero char at the end of string - const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { - std.debug.print("{any}\n", .{e}); - std.debug.dumpHex(s); - break :blk 0; - }; - // FIXME: this algorithm isn't working to find addresses "inside" symbols - if (v > 0) { - const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { - fn inner(a: u64, sym: SymbolRange) std.math.Order { - return std.math.order(a, sym.start); - } - }.inner); - if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { - // if (idx > 0) - // idx -= 1; - const d = v - symbols[idx].start; - if (d > 0) - return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); - return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); - } - } +) !SectionInfo { + var info: SectionInfo = .{ .all = try .initCapacity(gpa, 8) }; + var it = header.iterateSectionHeaders(reader); + var idx: u32 = 0; + while (try it.next()) |s| { + defer idx += 1; + try info.all.append(gpa, s); + if (idx == header.shstrndx) { + std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); + info.shstrtab = s; + } + switch (s.sh_type) { + std.elf.SHT_SYMTAB => info.symtab = s, + std.elf.SHT_DYNSYM => info.dynsym = s, + else => {}, } } + std.mem.sort(std.elf.Elf64_Shdr, info.all.items, {}, struct { + fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { + return x.sh_addr < y.sh_addr; + } + }.inner); + return info; +} +fn findStrtab(sections: []const std.elf.Elf64_Shdr, shstrtab_data: []const u8) ?std.elf.Elf64_Shdr { + for (sections) |s| { + if (s.sh_type == std.elf.SHT_STRTAB and + std.mem.eql(u8, ".strtab", std.mem.sliceTo(shstrtab_data[s.sh_name..], 0))) + return s; + } return null; } +// --- Symbol collection --- + +fn collectSymbols( + gpa: std.mem.Allocator, + header: std.elf.Header, + reader: *std.Io.File.Reader, + strtab_data: ?[]const u8, + sections: SectionInfo, +) !std.ArrayList(SymbolRange) { + var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); + if (strtab_data) |data| { + if (sections.symtab) |st| + try collectSymbolsFrom(gpa, header, reader, st, data, &syms); + // TODO: dynsym should technically use .dynstr, not .strtab + if (sections.dynsym) |ds| + try collectSymbolsFrom(gpa, header, reader, ds, data, &syms); + } + std.mem.sort(SymbolRange, syms.items, {}, struct { + fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { + return x.start < y.start; + } + }.inner); + return syms; +} + +fn collectSymbolsFrom( + gpa: std.mem.Allocator, + header: std.elf.Header, + reader: *std.Io.File.Reader, + section: std.elf.Elf64_Shdr, + strtab_data: []const u8, + syms: *std.ArrayList(SymbolRange), +) !void { + var it = iterateSymbols(header, reader, section); + while (try it.next()) |s| { + const name = std.mem.sliceTo(strtab_data[s.st_name..], 0); + const owned = try gpa.alloc(u8, name.len); + @memcpy(owned, name); + try syms.append(gpa, .{ + .start = s.st_value, + .end = s.st_value + s.st_size, + .name = owned, + .kind = s.st_info & 0xf, + }); + } +} + +// --- Rendering --- + fn printDisassembly( gpa: std.mem.Allocator, bw: *std.Io.Writer, @@ -394,41 +433,40 @@ fn printDisassembly( if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { try term.setColor(.blue); - try bw.print("\n{x:0>16} {s}:\n", .{ - addr, - symbols[idx].name, - }); + try bw.print("\n{x:0>16} {s}:\n", .{ addr, symbols[idx].name }); try term.setColor(.reset); } + try term.setColor(.dim); try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); try term.setColor(.reset); - // if(instr.detail.) + try term.setColor(.bright_green); try bw.print("{s} ", .{instr.mnemonic}); try term.setColor(.reset); - const mnemonic_strlen: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?); - const mnemonic_pad: u64 = 5; - for (0..(if (mnemonic_pad >= mnemonic_strlen) mnemonic_pad - mnemonic_strlen else 0)) |_| { + const mnemonic_len: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?); + const pad: u64 = 5; + for (0..(if (pad >= mnemonic_len) pad - mnemonic_len else 0)) |_| { try bw.printAsciiChar(' ', .{}); } + try bw.print("{s}", .{instr.op_str}); - const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols); - defer { - if (asm_comment != null) - gpa.free(asm_comment.?); - } - if (asm_comment != null and asm_comment.?.len > 0) { - try term.setColor(.blue); - try bw.print(" <{s}>", .{asm_comment.?}); + + if (try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols)) |comment| { + defer gpa.free(comment); + if (comment.len > 0) { + try term.setColor(.blue); + try bw.print(" <{s}>", .{comment}); + } } + try bw.print("\n", .{}); try term.setColor(.reset); } } -/// Prints a hexadecimal view of the bytes, returning any error that occurs. +/// Prints a hexadecimal view of the bytes. pub fn printHexdump( _: type, bw: *std.Io.Writer, @@ -436,19 +474,14 @@ pub fn printHexdump( bytes: []const u8, offset: u64, ) !void { - // @breakpoint(); const nbytes = 16; var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); while (chunks.next()) |window| { - // 1. Print the address. const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; try term.setColor(.dim); - // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more. - // Also, make sure all lines are aligned by padding the address. try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); try term.setColor(.reset); - // 2. Print the bytes. for (window, 0..) |byte, index| { try bw.print("{X:0>2} ", .{byte}); if (index == 7) try bw.writeByte(' '); @@ -461,28 +494,70 @@ pub fn printHexdump( } const window_bytes: []const u8 = @ptrCast(@alignCast(window)); - - // 3. Print the characters. for (window_bytes) |byte| { if (std.ascii.isPrint(byte)) { try bw.writeByte(byte); - } else { + } else switch (byte) { + '\n' => try bw.writeAll("␊"), + '\r' => try bw.writeAll("␍"), + '\t' => try bw.writeAll("␉"), + else => try bw.writeByte('.'), + } + } + try bw.writeByte('\n'); + } +} - // Let's print some common control codes as graphical Unicode symbols. - // We don't want to do this for all control codes because most control codes apart from - // the ones that Zig has escape sequences for are likely not very useful to print as symbols. - switch (byte) { - '\n' => try bw.writeAll("␊"), - '\r' => try bw.writeAll("␍"), - '\t' => try bw.writeAll("␉"), - else => try bw.writeByte('.'), +// --- Helpers --- + +fn initCapstone(header: std.elf.Header) usize { + var handle: usize = undefined; + const opts: struct { arch: u64, mode: u64 } = switch (header.machine) { + .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 }, + .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = if (header.is_64) cs.CS_MODE_64 else cs.CS_MODE_32 }, + else => { + std.debug.print("found machine: {any}\n", .{header.machine}); + @panic("unhandled arch"); + }, + }; + std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK); + return handle; +} + +fn allocComment( + gpa: std.mem.Allocator, + code: []u8, + symbols: []SymbolRange, +) !?[]u8 { + var iter = std.mem.splitAny(u8, code, " \t[],+-"); + while (iter.next()) |s| { + if (std.mem.startsWith(u8, s, "0x")) { + const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { + std.debug.print("{any}\n", .{e}); + std.debug.dumpHex(s); + break :blk 0; + }; + // FIXME: this algorithm isn't working to find addresses "inside" symbols + if (v > 0) { + const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { + fn inner(a: u64, sym: SymbolRange) std.math.Order { + return std.math.order(a, sym.start); + } + }.inner); + if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { + const d = v - symbols[idx].start; + if (d > 0) + return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); + return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); } } } - try bw.writeByte('\n'); } + return null; } +// --- Types --- + const SymbolRange = struct { start: u64, end: u64, @@ -510,7 +585,7 @@ const SymbolIterator = struct { pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { defer it.index += 1; - + // TODO: handle 32-bit symbols (Elf32_Sym) — currently both branches use the same size const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); const offset = it.symtab.sh_offset + size * it.index; |
