summaryrefslogtreecommitdiff
path: root/src/elfo.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/elfo.zig')
-rw-r--r--src/elfo.zig721
1 files changed, 398 insertions, 323 deletions
diff --git a/src/elfo.zig b/src/elfo.zig
index e07620c..73b666e 100644
--- a/src/elfo.zig
+++ b/src/elfo.zig
@@ -1,7 +1,5 @@
const std = @import("std");
const cs = @import("capstone");
-const hooks = @import("hooks.zig");
-const meta = @import("meta");
pub fn main(init: std.process.Init.Minimal) !void {
var debug_alloc: std.heap.DebugAllocator(.{}) = .init;
@@ -18,36 +16,33 @@ pub fn main(init: std.process.Init.Minimal) !void {
defer args.deinit();
_ = args.next(); // skip argv[0]
+ var path: []const u8 = "./study-samples/split";
+ var symbol_filter: ?[]const u8 = null;
+ var compare = false;
+ while (args.next()) |arg| {
+ if (std.mem.eql(u8, arg, "-s")) {
+ symbol_filter = args.next();
+ } else if (std.mem.eql(u8, arg, "--compare")) {
+ compare = true;
+ } else {
+ path = arg;
+ }
+ }
+
var buffer: [64]u8 = undefined;
const stderr = try io.lockStderr(&buffer, null);
- // const hook = hooks.init(
- // if (meta.gdb) .breakpoint else .none,
- // // TODO: this needs to be comptime, it'd be cool to don't need that
- // std.heap.page_allocator,
- // false,
- // );
-
- // const mem_config: cs.cs_opt_mem = .{
- // .malloc = @ptrCast(&hook.malloc),
- // .free = @ptrCast(&hook.free),
- // .calloc = @ptrCast(&hook.calloc),
- // .realloc = @ptrCast(&hook.realloc),
- // // TODO: this function needs to be properly implemented.
- // .vsnprintf = @ptrCast(&hook.vsnprintf),
- // };
- // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK);
-
- try printElf(
- alloc,
- io,
- args.next() orelse "./study-samples/split",
- stderr.terminal(),
- .{
- // .show_unaddressable_sections = true,
- // .skip_sections_content = true,
- },
- );
+ if (compare) {
+ try compareWithObjdump(alloc, io, path, stderr.terminal());
+ } else {
+ try printElf(
+ alloc,
+ io,
+ path,
+ stderr.terminal(),
+ .{ .symbol_filter = symbol_filter },
+ );
+ }
}
pub fn printElf(
@@ -58,200 +53,85 @@ pub fn printElf(
options: struct {
show_unaddressable_sections: bool = false,
skip_sections_content: bool = false,
+ symbol_filter: ?[]const u8 = null,
},
) !void {
const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only });
const bw = term.writer;
defer bw.flush() catch {};
+
var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100);
defer gpa.free(buffer);
-
var reader = f.reader(io, buffer);
+
const header = try std.elf.Header.read(&reader.interface);
- var cs_handle: usize = undefined;
+ var cs_handle = initCapstone(header);
defer _ = cs.cs_close(@ptrCast(&cs_handle));
- const cs_opts: struct { arch: u64, mode: u64 } = switch (header.machine) {
- .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 },
- // The arm mode isn't working with symbols on the .text section correctly
- .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = if (header.is_64) cs.CS_MODE_64 else cs.CS_MODE_32 },
- else => {
- std.debug.print("found machine: {any}\n", .{header.machine});
- @panic("unhandled arch");
- },
- };
-
- std.debug.assert(cs.cs_open(@intCast(cs_opts.arch), @intCast(cs_opts.mode), @ptrCast(&cs_handle)) == cs.CS_ERR_OK);
- const shstrtab = blk: {
- var section_it = header.iterateSectionHeaders(&reader);
- var section_idx: u32 = 0;
- while (try section_it.next()) |s| {
- defer section_idx += 1;
- if (section_idx == header.shstrndx) {
- std.debug.assert(s.sh_type == std.elf.SHT_STRTAB);
- break :blk s;
- }
- }
- break :blk null;
- };
-
- const elf_shstrtab_slice = blk: {
- if (shstrtab == null)
- break :blk null;
-
- try reader.seekTo(shstrtab.?.sh_offset);
- const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size);
- break :blk slice;
- };
- defer {
- if (elf_shstrtab_slice != null)
- gpa.free(elf_shstrtab_slice.?);
- }
+ // Single-pass: collect all section headers and find key sections by type/index
+ var sections = try collectSections(header, &reader, gpa);
+ defer sections.deinit(gpa);
- const strtab = blk: {
- if (elf_shstrtab_slice == null)
- break :blk null;
- var section_it = header.iterateSectionHeaders(&reader);
- while (try section_it.next()) |s| {
- if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql(
- u8,
- ".strtab",
- std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0),
- ))
- // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0)
- break :blk s;
- }
- break :blk null;
- };
+ // Load section header string table
+ const shstrtab_data: ?[]u8 = if (sections.shstrtab) |s| blk: {
+ try reader.seekTo(s.sh_offset);
+ break :blk try reader.interface.readAlloc(gpa, s.sh_size);
+ } else null;
+ defer if (shstrtab_data) |d| gpa.free(d);
- const elf_strtab_slice = blk: {
- if (strtab == null)
- break :blk null;
+ // Find .strtab by name (requires shstrtab_data)
+ const strtab = if (shstrtab_data) |data|
+ findStrtab(sections.all.items, data)
+ else
+ null;
- try reader.seekTo(strtab.?.sh_offset);
- const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size);
- break :blk slice;
- };
+ // Load symbol string table
+ const strtab_data: ?[]u8 = if (strtab) |s| blk: {
+ try reader.seekTo(s.sh_offset);
+ break :blk try reader.interface.readAlloc(gpa, s.sh_size);
+ } else null;
+ defer if (strtab_data) |d| gpa.free(d);
+ // Collect symbols from symtab and dynsym
+ var symbols = try collectSymbols(gpa, header, &reader, strtab_data, sections);
defer {
- if (elf_strtab_slice != null)
- gpa.free(elf_strtab_slice.?);
+ for (symbols.items) |sym| gpa.free(sym.name);
+ symbols.deinit(gpa);
}
- var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8);
- {
- if (elf_shstrtab_slice != null) {
- var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0);
- while (str_it.next()) |str| {
- const owned_str = try gpa.alloc(u8, str.len);
- @memcpy(owned_str, str);
- try strs.append(gpa, owned_str);
- }
- }
- }
- defer {
- for (strs.items) |s| {
- gpa.free(s);
+ // When filtering by symbol, find the target and only render its disassembly
+ const filter_sym: ?SymbolRange = if (options.symbol_filter) |name| blk: {
+ for (symbols.items) |sym| {
+ if (std.mem.eql(u8, sym.name, name)) break :blk sym;
}
- strs.deinit(gpa);
- }
+ break :blk null;
+ } else null;
- var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8);
- {
- var section_it = header.iterateSectionHeaders(&reader);
- while (try section_it.next()) |section| {
- try sections.append(gpa, section);
+ if (options.symbol_filter == null) {
+ for (symbols.items) |sym| {
+ if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0)
+ try bw.print("{x}-{x} {s}\n", .{ sym.start, sym.end, sym.name });
}
- std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct {
- pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool {
- return x.sh_addr < y.sh_addr;
- }
- }.inner);
}
- defer sections.deinit(gpa);
- const symtab = blk: {
- var section_it = header.iterateSectionHeaders(&reader);
- while (try section_it.next()) |s| {
- if (s.sh_type == std.elf.SHT_SYMTAB) {
+ // Render sections
+ for (sections.all.items) |section| {
+ if (section.sh_size == 0) continue;
+ const addressable = section.sh_addr > 0;
+ if (!addressable and !options.show_unaddressable_sections) continue;
- // x64: symtab: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4256, .sh_size = 1680, .sh_link = 27, .sh_info = 45, .sh_addralign = 8, .sh_entsize = 24 }
- // arm32: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4264, .sh_size = 1856, .sh_link = 27, .sh_info = 87, .sh_addralign = 4, .sh_entsize = 16 }
- // try bw.print("symtab: {any}\n", .{s});
- break :blk s;
- }
- }
- break :blk null;
- };
+ const is_exec = addressable and section.sh_type == std.elf.SHT_PROGBITS and
+ (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0;
- const dynsym = blk: {
- var section_it = header.iterateSectionHeaders(&reader);
- while (try section_it.next()) |s| {
- if (s.sh_type == std.elf.SHT_DYNSYM) {
- // try bw.print("sym: {any}\n", .{s});
- break :blk s;
- }
+ // When filtering by symbol, skip sections that don't contain it
+ if (filter_sym) |fsym| {
+ if (!is_exec) continue;
+ const sec_end = section.sh_addr + section.sh_size;
+ if (fsym.start < section.sh_addr or fsym.start >= sec_end) continue;
}
- break :blk null;
- };
- // try bw.print("dynsym: {any}\n", .{dynsym});
- var symbols_index = blk: {
- var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8);
- if (symtab != null) {
- var sym_it = iterateSymbols(header, &reader, symtab.?);
- while (try sym_it.next()) |s| {
- const t = s.st_info & 0xf;
- const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0);
- const owned_name = try gpa.alloc(u8, name.len);
- @memcpy(owned_name, name);
- try syms.append(gpa, .{
- .start = s.st_value,
- .end = s.st_value + s.st_size,
- .name = owned_name,
- .kind = t,
- });
- }
- }
-
- // the check on elf_strtab_slice might not be necessary
- if (dynsym != null and elf_strtab_slice != null) {
- var sym_it = iterateSymbols(header, &reader, dynsym.?);
- while (try sym_it.next()) |s| {
- const t = s.st_info & 0xf;
- const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0);
- const owned_name = try gpa.alloc(u8, name.len);
- @memcpy(owned_name, name);
- try syms.append(gpa, .{
- .start = s.st_value,
- .end = s.st_value + s.st_size,
- .name = owned_name,
- .kind = t,
- });
- }
- }
- std.mem.sort(SymbolRange, syms.items, {}, struct {
- fn inner(_: void, x: SymbolRange, y: SymbolRange) bool {
- return x.start < y.start;
- }
- }.inner);
- break :blk syms;
- };
- defer {
- for (symbols_index.items) |sym| {
- gpa.free(sym.name);
- }
- symbols_index.deinit(gpa);
- }
-
- for (symbols_index.items) |sym| {
- if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0)
- try bw.print("{x}-{x} {s}\n", .{ sym.start, sym.end, sym.name });
- }
-
- for (sections.items) |section| {
- if (section.sh_size > 0 and section.sh_addr > 0) {
+ if (options.symbol_filter == null) {
try term.setColor(.reset);
try term.setColor(.dim);
try bw.print("\n{x}-{x} (t: {x}) -- ", .{
@@ -259,124 +139,283 @@ pub fn printElf(
section.sh_addr + section.sh_size,
section.sh_type,
});
-
- try term.setColor(.bright_green);
- if (elf_shstrtab_slice != null)
- try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)});
+ try term.setColor(if (addressable) .bright_green else .bright_cyan);
+ if (shstrtab_data) |data|
+ try bw.print("{s}", .{std.mem.sliceTo(data[section.sh_name..], 0)});
try bw.print("\n", .{});
try term.setColor(.reset);
+ }
- // --
- try reader.seekTo(section.sh_offset);
+ try reader.seekTo(section.sh_offset);
+ if (buffer.len < section.sh_size) {
+ buffer = try gpa.realloc(buffer, section.sh_size);
+ reader = f.reader(io, buffer);
+ }
+ const section_slice = reader.interface.take(section.sh_size) catch |e| switch (e) {
+ error.EndOfStream => {
+ if (addressable) try bw.print("failed\n", .{});
+ continue;
+ },
+ error.ReadFailed => unreachable,
+ };
- if (buffer.len < section.sh_size) {
- buffer = try gpa.realloc(buffer, section.sh_size);
- reader = f.reader(io, buffer);
- }
- const section_slice = reader.interface.take(section.sh_size) catch |e| blk: {
- switch (e) {
- error.EndOfStream => {
- try bw.print("failed\n", .{});
- break :blk null;
- },
- error.ReadFailed => unreachable,
- }
- };
- // TODO: this heuristic is probably wrong
- if (section_slice != null and !options.skip_sections_content) {
- if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) {
- const instrs: []cs.cs_insn = blk: {
- var insn: [*]cs.cs_insn = undefined;
- // TODO: use iter API
- // https://www.capstone-engine.org/iteration.html
- // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn));
- const count = cs.cs_disasm(cs_handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn));
- break :blk insn[0..count];
- };
+ if (options.skip_sections_content) continue;
- try printDisassembly(gpa, bw, term, instrs, symbols_index.items);
- } else {
- try printHexdump(u64, bw, term, section_slice.?, section.sh_addr);
+ if (is_exec) {
+ var insn: [*]cs.cs_insn = undefined;
+ // TODO: use iter API https://www.capstone-engine.org/iteration.html
+ const count = cs.cs_disasm(cs_handle, section_slice.ptr, section_slice.len, section.sh_addr, 0, @ptrCast(&insn));
+ const instrs = insn[0..count];
+
+ if (filter_sym) |fsym| {
+ // Find instruction range within the symbol
+ var start_idx: usize = 0;
+ var end_idx: usize = instrs.len;
+ for (instrs, 0..) |instr, i| {
+ if (instr.address >= fsym.start and start_idx == 0) start_idx = i;
+ if (instr.address >= fsym.end and fsym.end > fsym.start) {
+ end_idx = i;
+ break;
+ }
}
+ try printDisassembly(gpa, bw, term, instrs[start_idx..end_idx], symbols.items);
+ } else {
+ try printDisassembly(gpa, bw, term, instrs, symbols.items);
}
+ } else {
+ try printHexdump(u64, bw, term, section_slice, section.sh_addr);
}
}
+}
- if (options.show_unaddressable_sections) {
- for (sections.items) |section| {
- if (section.sh_size > 0 and section.sh_addr == 0) {
- try term.setColor(.reset);
- try term.setColor(.dim);
- try bw.print("{x}-{x} (t: {x}) -- ", .{
- section.sh_addr,
- section.sh_addr + section.sh_size,
- section.sh_type,
- });
- try term.setColor(.bright_cyan);
- if (elf_shstrtab_slice != null)
- try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)});
- try bw.print("\n", .{});
- try term.setColor(.reset);
- // --
+fn compareWithObjdump(
+ gpa: std.mem.Allocator,
+ io: std.Io,
+ path: []const u8,
+ term: std.Io.Terminal,
+) !void {
+ const bw = term.writer;
+ defer bw.flush() catch {};
- try reader.seekTo(section.sh_offset);
+ var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100);
+ defer gpa.free(buffer);
- if (buffer.len < section.sh_size) {
- buffer = try gpa.realloc(buffer, section.sh_size);
- reader = f.reader(io, buffer);
- }
- const section_slice = reader.interface.take(section.sh_size) catch |e| blk: {
- switch (e) {
- error.EndOfStream => {
- break :blk null;
- },
- error.ReadFailed => unreachable,
- }
- };
- if (section_slice != null and !options.skip_sections_content) {
- try printHexdump(u64, bw, term, section_slice.?, section.sh_addr);
+ const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only });
+ var reader = f.reader(io, buffer);
+ const header = try std.elf.Header.read(&reader.interface);
+
+ var cs_handle = initCapstone(header);
+ defer _ = cs.cs_close(@ptrCast(&cs_handle));
+
+ var sections = try collectSections(header, &reader, gpa);
+ defer sections.deinit(gpa);
+
+ const shstrtab_data: ?[]u8 = if (sections.shstrtab) |s| blk: {
+ try reader.seekTo(s.sh_offset);
+ break :blk try reader.interface.readAlloc(gpa, s.sh_size);
+ } else null;
+ defer if (shstrtab_data) |d| gpa.free(d);
+
+ const strtab = if (shstrtab_data) |data| findStrtab(sections.all.items, data) else null;
+ const strtab_data: ?[]u8 = if (strtab) |s| blk: {
+ try reader.seekTo(s.sh_offset);
+ break :blk try reader.interface.readAlloc(gpa, s.sh_size);
+ } else null;
+ defer if (strtab_data) |d| gpa.free(d);
+
+ var symbols = try collectSymbols(gpa, header, &reader, strtab_data, sections);
+ defer {
+ for (symbols.items) |sym| gpa.free(sym.name);
+ symbols.deinit(gpa);
+ }
+
+ // For each function symbol with nonzero size, show elfo vs objdump
+ for (symbols.items) |sym| {
+ if (sym.kind != std.elf.STT_FUNC or sym.name.len == 0 or sym.start == sym.end) continue;
+
+ // Find the section containing this symbol
+ const section = blk: {
+ for (sections.all.items) |s| {
+ const is_exec = s.sh_addr > 0 and s.sh_type == std.elf.SHT_PROGBITS and
+ (s.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0;
+ if (is_exec and sym.start >= s.sh_addr and sym.start < s.sh_addr + s.sh_size)
+ break :blk s;
+ }
+ continue;
+ };
+
+ // Header
+ try term.setColor(.bright_green);
+ try bw.writeAll("\n============================================================\n");
+ try bw.print(" {s} ({x:0>16} - {x:0>16})\n", .{ sym.name, sym.start, sym.end });
+ try bw.writeAll("============================================================\n");
+ try term.setColor(.reset);
+
+ // --- elfo output ---
+ try term.setColor(.blue);
+ try bw.print("--- elfo ---\n", .{});
+ try term.setColor(.reset);
+
+ try reader.seekTo(section.sh_offset);
+ if (buffer.len < section.sh_size) {
+ buffer = try gpa.realloc(buffer, section.sh_size);
+ reader = f.reader(io, buffer);
+ }
+ const section_slice = reader.interface.take(section.sh_size) catch |e| switch (e) {
+ error.EndOfStream => {
+ try bw.print("failed to read section\n", .{});
+ continue;
+ },
+ error.ReadFailed => unreachable,
+ };
+
+ {
+ var insn: [*]cs.cs_insn = undefined;
+ const count = cs.cs_disasm(cs_handle, section_slice.ptr, section_slice.len, section.sh_addr, 0, @ptrCast(&insn));
+ const instrs = insn[0..count];
+
+ var start_idx: usize = 0;
+ var end_idx: usize = instrs.len;
+ for (instrs, 0..) |instr, i| {
+ if (instr.address >= sym.start and start_idx == 0) start_idx = i;
+ if (instr.address >= sym.end) {
+ end_idx = i;
+ break;
}
}
+ try printDisassembly(gpa, bw, term, instrs[start_idx..end_idx], symbols.items);
}
+ bw.flush() catch {};
+
+ // --- objdump output ---
+ try term.setColor(.blue);
+ try bw.print("\n--- objdump ---\n", .{});
+ try term.setColor(.reset);
+ bw.flush() catch {};
+
+ const start_addr = try std.fmt.allocPrint(gpa, "0x{x}", .{sym.start});
+ defer gpa.free(start_addr);
+ const stop_addr = try std.fmt.allocPrint(gpa, "0x{x}", .{sym.end});
+ defer gpa.free(stop_addr);
+
+ const result = std.process.run(gpa, io, .{
+ .argv = &.{
+ "objdump", "-d", "-M", "intel", "--no-show-raw-insn",
+ "--start-address", start_addr,
+ "--stop-address", stop_addr,
+ path,
+ },
+ }) catch |e| {
+ try bw.print("failed to run objdump: {any}\n", .{e});
+ continue;
+ };
+ defer gpa.free(result.stdout);
+ defer gpa.free(result.stderr);
+
+ try bw.writeAll(result.stdout);
}
}
-fn allocComment(
+// --- Section collection ---
+
+const SectionInfo = struct {
+ all: std.ArrayList(std.elf.Elf64_Shdr),
+ shstrtab: ?std.elf.Elf64_Shdr = null,
+ symtab: ?std.elf.Elf64_Shdr = null,
+ dynsym: ?std.elf.Elf64_Shdr = null,
+
+ fn deinit(self: *SectionInfo, gpa: std.mem.Allocator) void {
+ self.all.deinit(gpa);
+ }
+};
+
+fn collectSections(
+ header: std.elf.Header,
+ reader: *std.Io.File.Reader,
gpa: std.mem.Allocator,
- code: []u8,
- symbols: []SymbolRange,
-) !?[]u8 {
- var iter = std.mem.splitAny(u8, code, " \t[],+-");
- while (iter.next()) |s| {
- if (std.mem.startsWith(u8, s, "0x")) {
- // todo split at the zero char at the end of string
- const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: {
- std.debug.print("{any}\n", .{e});
- std.debug.dumpHex(s);
- break :blk 0;
- };
- // FIXME: this algorithm isn't working to find addresses "inside" symbols
- if (v > 0) {
- const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct {
- fn inner(a: u64, sym: SymbolRange) std.math.Order {
- return std.math.order(a, sym.start);
- }
- }.inner);
- if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) {
- // if (idx > 0)
- // idx -= 1;
- const d = v - symbols[idx].start;
- if (d > 0)
- return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d });
- return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name});
- }
- }
+) !SectionInfo {
+ var info: SectionInfo = .{ .all = try .initCapacity(gpa, 8) };
+ var it = header.iterateSectionHeaders(reader);
+ var idx: u32 = 0;
+ while (try it.next()) |s| {
+ defer idx += 1;
+ try info.all.append(gpa, s);
+ if (idx == header.shstrndx) {
+ std.debug.assert(s.sh_type == std.elf.SHT_STRTAB);
+ info.shstrtab = s;
+ }
+ switch (s.sh_type) {
+ std.elf.SHT_SYMTAB => info.symtab = s,
+ std.elf.SHT_DYNSYM => info.dynsym = s,
+ else => {},
}
}
+ std.mem.sort(std.elf.Elf64_Shdr, info.all.items, {}, struct {
+ fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool {
+ return x.sh_addr < y.sh_addr;
+ }
+ }.inner);
+ return info;
+}
+fn findStrtab(sections: []const std.elf.Elf64_Shdr, shstrtab_data: []const u8) ?std.elf.Elf64_Shdr {
+ for (sections) |s| {
+ if (s.sh_type == std.elf.SHT_STRTAB and
+ std.mem.eql(u8, ".strtab", std.mem.sliceTo(shstrtab_data[s.sh_name..], 0)))
+ return s;
+ }
return null;
}
+// --- Symbol collection ---
+
+fn collectSymbols(
+ gpa: std.mem.Allocator,
+ header: std.elf.Header,
+ reader: *std.Io.File.Reader,
+ strtab_data: ?[]const u8,
+ sections: SectionInfo,
+) !std.ArrayList(SymbolRange) {
+ var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8);
+ if (strtab_data) |data| {
+ if (sections.symtab) |st|
+ try collectSymbolsFrom(gpa, header, reader, st, data, &syms);
+ // TODO: dynsym should technically use .dynstr, not .strtab
+ if (sections.dynsym) |ds|
+ try collectSymbolsFrom(gpa, header, reader, ds, data, &syms);
+ }
+ std.mem.sort(SymbolRange, syms.items, {}, struct {
+ fn inner(_: void, x: SymbolRange, y: SymbolRange) bool {
+ return x.start < y.start;
+ }
+ }.inner);
+ return syms;
+}
+
+fn collectSymbolsFrom(
+ gpa: std.mem.Allocator,
+ header: std.elf.Header,
+ reader: *std.Io.File.Reader,
+ section: std.elf.Elf64_Shdr,
+ strtab_data: []const u8,
+ syms: *std.ArrayList(SymbolRange),
+) !void {
+ var it = iterateSymbols(header, reader, section);
+ while (try it.next()) |s| {
+ const name = std.mem.sliceTo(strtab_data[s.st_name..], 0);
+ const owned = try gpa.alloc(u8, name.len);
+ @memcpy(owned, name);
+ try syms.append(gpa, .{
+ .start = s.st_value,
+ .end = s.st_value + s.st_size,
+ .name = owned,
+ .kind = s.st_info & 0xf,
+ });
+ }
+}
+
+// --- Rendering ---
+
fn printDisassembly(
gpa: std.mem.Allocator,
bw: *std.Io.Writer,
@@ -394,41 +433,40 @@ fn printDisassembly(
if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) {
try term.setColor(.blue);
- try bw.print("\n{x:0>16} {s}:\n", .{
- addr,
- symbols[idx].name,
- });
+ try bw.print("\n{x:0>16} {s}:\n", .{ addr, symbols[idx].name });
try term.setColor(.reset);
}
+
try term.setColor(.dim);
try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 });
try term.setColor(.reset);
- // if(instr.detail.)
+
try term.setColor(.bright_green);
try bw.print("{s} ", .{instr.mnemonic});
try term.setColor(.reset);
- const mnemonic_strlen: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?);
- const mnemonic_pad: u64 = 5;
- for (0..(if (mnemonic_pad >= mnemonic_strlen) mnemonic_pad - mnemonic_strlen else 0)) |_| {
+ const mnemonic_len: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?);
+ const pad: u64 = 5;
+ for (0..(if (pad >= mnemonic_len) pad - mnemonic_len else 0)) |_| {
try bw.printAsciiChar(' ', .{});
}
+
try bw.print("{s}", .{instr.op_str});
- const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols);
- defer {
- if (asm_comment != null)
- gpa.free(asm_comment.?);
- }
- if (asm_comment != null and asm_comment.?.len > 0) {
- try term.setColor(.blue);
- try bw.print(" <{s}>", .{asm_comment.?});
+
+ if (try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols)) |comment| {
+ defer gpa.free(comment);
+ if (comment.len > 0) {
+ try term.setColor(.blue);
+ try bw.print(" <{s}>", .{comment});
+ }
}
+
try bw.print("\n", .{});
try term.setColor(.reset);
}
}
-/// Prints a hexadecimal view of the bytes, returning any error that occurs.
+/// Prints a hexadecimal view of the bytes.
pub fn printHexdump(
_: type,
bw: *std.Io.Writer,
@@ -436,19 +474,14 @@ pub fn printHexdump(
bytes: []const u8,
offset: u64,
) !void {
- // @breakpoint();
const nbytes = 16;
var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes);
while (chunks.next()) |window| {
- // 1. Print the address.
const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset;
try term.setColor(.dim);
- // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more.
- // Also, make sure all lines are aligned by padding the address.
try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 });
try term.setColor(.reset);
- // 2. Print the bytes.
for (window, 0..) |byte, index| {
try bw.print("{X:0>2} ", .{byte});
if (index == 7) try bw.writeByte(' ');
@@ -461,28 +494,70 @@ pub fn printHexdump(
}
const window_bytes: []const u8 = @ptrCast(@alignCast(window));
-
- // 3. Print the characters.
for (window_bytes) |byte| {
if (std.ascii.isPrint(byte)) {
try bw.writeByte(byte);
- } else {
+ } else switch (byte) {
+ '\n' => try bw.writeAll("␊"),
+ '\r' => try bw.writeAll("␍"),
+ '\t' => try bw.writeAll("␉"),
+ else => try bw.writeByte('.'),
+ }
+ }
+ try bw.writeByte('\n');
+ }
+}
- // Let's print some common control codes as graphical Unicode symbols.
- // We don't want to do this for all control codes because most control codes apart from
- // the ones that Zig has escape sequences for are likely not very useful to print as symbols.
- switch (byte) {
- '\n' => try bw.writeAll("␊"),
- '\r' => try bw.writeAll("␍"),
- '\t' => try bw.writeAll("␉"),
- else => try bw.writeByte('.'),
+// --- Helpers ---
+
+fn initCapstone(header: std.elf.Header) usize {
+ var handle: usize = undefined;
+ const opts: struct { arch: u64, mode: u64 } = switch (header.machine) {
+ .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 },
+ .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = if (header.is_64) cs.CS_MODE_64 else cs.CS_MODE_32 },
+ else => {
+ std.debug.print("found machine: {any}\n", .{header.machine});
+ @panic("unhandled arch");
+ },
+ };
+ std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK);
+ return handle;
+}
+
+fn allocComment(
+ gpa: std.mem.Allocator,
+ code: []u8,
+ symbols: []SymbolRange,
+) !?[]u8 {
+ var iter = std.mem.splitAny(u8, code, " \t[],+-");
+ while (iter.next()) |s| {
+ if (std.mem.startsWith(u8, s, "0x")) {
+ const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: {
+ std.debug.print("{any}\n", .{e});
+ std.debug.dumpHex(s);
+ break :blk 0;
+ };
+ // FIXME: this algorithm isn't working to find addresses "inside" symbols
+ if (v > 0) {
+ const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct {
+ fn inner(a: u64, sym: SymbolRange) std.math.Order {
+ return std.math.order(a, sym.start);
+ }
+ }.inner);
+ if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) {
+ const d = v - symbols[idx].start;
+ if (d > 0)
+ return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d });
+ return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name});
}
}
}
- try bw.writeByte('\n');
}
+ return null;
}
+// --- Types ---
+
const SymbolRange = struct {
start: u64,
end: u64,
@@ -510,7 +585,7 @@ const SymbolIterator = struct {
pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym {
defer it.index += 1;
-
+ // TODO: handle 32-bit symbols (Elf32_Sym) — currently both branches use the same size
const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym);
const offset = it.symtab.sh_offset + size * it.index;