const std = @import("std"); const cs = @import("capstone"); pub fn main(init: std.process.Init.Minimal) !void { var debug_alloc: std.heap.DebugAllocator(.{}) = .init; const alloc = debug_alloc.allocator(); var threaded: std.Io.Threaded = .init(alloc, .{ .argv0 = .init(init.args), .environ = init.environ, }); defer threaded.deinit(); const io = threaded.io(); var args = try init.args.iterateAllocator(alloc); defer args.deinit(); _ = args.next(); // skip argv[0] var path: []const u8 = "./study-samples/split"; var symbol_filter: ?[]const u8 = null; var compare = false; while (args.next()) |arg| { if (std.mem.eql(u8, arg, "-s")) { symbol_filter = args.next(); } else if (std.mem.eql(u8, arg, "--compare")) { compare = true; } else { path = arg; } } var buffer: [64]u8 = undefined; const stderr = try io.lockStderr(&buffer, null); if (compare) { try compareWithObjdump(alloc, io, path, stderr.terminal()); } else { try printElf( alloc, io, path, stderr.terminal(), .{ .symbol_filter = symbol_filter }, ); } } pub fn printElf( gpa: std.mem.Allocator, io: std.Io, path: []const u8, term: std.Io.Terminal, options: struct { show_unaddressable_sections: bool = false, skip_sections_content: bool = false, symbol_filter: ?[]const u8 = null, }, ) !void { const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); const bw = term.writer; defer bw.flush() catch {}; var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); defer gpa.free(buffer); var reader = f.reader(io, buffer); const header = try std.elf.Header.read(&reader.interface); var cs_handle = initCapstone(header); defer _ = cs.cs_close(@ptrCast(&cs_handle)); // Single-pass: collect all section headers and find key sections by type/index var sections = try collectSections(header, &reader, gpa); defer sections.deinit(gpa); // Load section header string table const shstrtab_data: ?[]u8 = if (sections.shstrtab) |s| blk: { try reader.seekTo(s.sh_offset); break :blk try reader.interface.readAlloc(gpa, s.sh_size); } else null; defer if (shstrtab_data) |d| gpa.free(d); // Load .strtab (for .symtab symbols, resolved via sh_link) const strtab_data: ?[]u8 = if (sections.strtab) |s| blk: { try reader.seekTo(s.sh_offset); break :blk try reader.interface.readAlloc(gpa, s.sh_size); } else null; defer if (strtab_data) |d| gpa.free(d); // Load .dynstr (for .dynsym symbols, resolved via sh_link) const dynstr_data: ?[]u8 = if (sections.dynstr) |s| blk: { try reader.seekTo(s.sh_offset); break :blk try reader.interface.readAlloc(gpa, s.sh_size); } else null; defer if (dynstr_data) |d| gpa.free(d); // Collect symbols from symtab and dynsym with their respective string tables var symbols = try collectSymbols(gpa, header, &reader, strtab_data, dynstr_data, shstrtab_data, sections); defer { for (symbols.items) |sym| gpa.free(sym.name); symbols.deinit(gpa); } // When filtering by symbol, find the target and only render its disassembly const filter_sym: ?SymbolRange = if (options.symbol_filter) |name| blk: { for (symbols.items) |sym| { if (std.mem.eql(u8, sym.name, name)) break :blk sym; } break :blk null; } else null; if (options.symbol_filter == null) { for (symbols.items) |sym| { if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) try bw.print("{x}-{x} {s}\n", .{ sym.start, sym.end, sym.name }); } } // Render sections for (sections.all.items) |section| { if (section.sh_size == 0) continue; const addressable = section.sh_addr > 0; if (!addressable and !options.show_unaddressable_sections) continue; const is_exec = addressable and section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0; // When filtering by symbol, skip sections that don't contain it if (filter_sym) |fsym| { if (!is_exec) continue; const sec_end = section.sh_addr + section.sh_size; if (fsym.start < section.sh_addr or fsym.start >= sec_end) continue; } if (options.symbol_filter == null) { try term.setColor(.reset); try term.setColor(.dim); try bw.print("\n{x}-{x} (t: {x}) -- ", .{ section.sh_addr, section.sh_addr + section.sh_size, section.sh_type, }); try term.setColor(if (addressable) .bright_green else .bright_cyan); if (shstrtab_data) |data| try bw.print("{s}", .{std.mem.sliceTo(data[section.sh_name..], 0)}); try bw.print("\n", .{}); try term.setColor(.reset); } try reader.seekTo(section.sh_offset); if (buffer.len < section.sh_size) { buffer = try gpa.realloc(buffer, section.sh_size); reader = f.reader(io, buffer); } const section_slice = reader.interface.take(section.sh_size) catch |e| switch (e) { error.EndOfStream => { if (addressable) try bw.print("failed\n", .{}); continue; }, error.ReadFailed => unreachable, }; if (options.skip_sections_content) continue; if (is_exec) { var insn: [*]cs.cs_insn = undefined; // TODO: use iter API https://www.capstone-engine.org/iteration.html const count = cs.cs_disasm(cs_handle, section_slice.ptr, section_slice.len, section.sh_addr, 0, @ptrCast(&insn)); const instrs = insn[0..count]; if (filter_sym) |fsym| { // Find instruction range within the symbol var start_idx: usize = 0; var end_idx: usize = instrs.len; for (instrs, 0..) |instr, i| { if (instr.address >= fsym.start and start_idx == 0) start_idx = i; if (instr.address >= fsym.end and fsym.end > fsym.start) { end_idx = i; break; } } try printDisassembly(gpa, bw, term, instrs[start_idx..end_idx], symbols.items); } else { try printDisassembly(gpa, bw, term, instrs, symbols.items); } } else { try printHexdump(u64, bw, term, section_slice, section.sh_addr); } } } fn compareWithObjdump( gpa: std.mem.Allocator, io: std.Io, path: []const u8, term: std.Io.Terminal, ) !void { const bw = term.writer; defer bw.flush() catch {}; var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); defer gpa.free(buffer); const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); var reader = f.reader(io, buffer); const header = try std.elf.Header.read(&reader.interface); var cs_handle = initCapstone(header); defer _ = cs.cs_close(@ptrCast(&cs_handle)); var sections = try collectSections(header, &reader, gpa); defer sections.deinit(gpa); const shstrtab_data: ?[]u8 = if (sections.shstrtab) |s| blk: { try reader.seekTo(s.sh_offset); break :blk try reader.interface.readAlloc(gpa, s.sh_size); } else null; defer if (shstrtab_data) |d| gpa.free(d); const strtab_data: ?[]u8 = if (sections.strtab) |s| blk: { try reader.seekTo(s.sh_offset); break :blk try reader.interface.readAlloc(gpa, s.sh_size); } else null; defer if (strtab_data) |d| gpa.free(d); const dynstr_data: ?[]u8 = if (sections.dynstr) |s| blk: { try reader.seekTo(s.sh_offset); break :blk try reader.interface.readAlloc(gpa, s.sh_size); } else null; defer if (dynstr_data) |d| gpa.free(d); var symbols = try collectSymbols(gpa, header, &reader, strtab_data, dynstr_data, shstrtab_data, sections); defer { for (symbols.items) |sym| gpa.free(sym.name); symbols.deinit(gpa); } // For each function symbol with nonzero size, show elfo vs objdump for (symbols.items) |sym| { if (sym.kind != std.elf.STT_FUNC or sym.name.len == 0 or sym.start == sym.end) continue; // Find the section containing this symbol const section = blk: { for (sections.all.items) |s| { const is_exec = s.sh_addr > 0 and s.sh_type == std.elf.SHT_PROGBITS and (s.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0; if (is_exec and sym.start >= s.sh_addr and sym.start < s.sh_addr + s.sh_size) break :blk s; } continue; }; // Header try term.setColor(.bright_green); try bw.writeAll("\n============================================================\n"); try bw.print(" {s} ({x:0>16} - {x:0>16})\n", .{ sym.name, sym.start, sym.end }); try bw.writeAll("============================================================\n"); try term.setColor(.reset); // --- elfo output --- try term.setColor(.blue); try bw.print("--- elfo ---\n", .{}); try term.setColor(.reset); try reader.seekTo(section.sh_offset); if (buffer.len < section.sh_size) { buffer = try gpa.realloc(buffer, section.sh_size); reader = f.reader(io, buffer); } const section_slice = reader.interface.take(section.sh_size) catch |e| switch (e) { error.EndOfStream => { try bw.print("failed to read section\n", .{}); continue; }, error.ReadFailed => unreachable, }; { var insn: [*]cs.cs_insn = undefined; const count = cs.cs_disasm(cs_handle, section_slice.ptr, section_slice.len, section.sh_addr, 0, @ptrCast(&insn)); const instrs = insn[0..count]; var start_idx: usize = 0; var end_idx: usize = instrs.len; for (instrs, 0..) |instr, i| { if (instr.address >= sym.start and start_idx == 0) start_idx = i; if (instr.address >= sym.end) { end_idx = i; break; } } try printDisassembly(gpa, bw, term, instrs[start_idx..end_idx], symbols.items); } bw.flush() catch {}; // --- objdump output --- try term.setColor(.blue); try bw.print("\n--- objdump ---\n", .{}); try term.setColor(.reset); bw.flush() catch {}; const start_addr = try std.fmt.allocPrint(gpa, "0x{x}", .{sym.start}); defer gpa.free(start_addr); const stop_addr = try std.fmt.allocPrint(gpa, "0x{x}", .{sym.end}); defer gpa.free(stop_addr); const result = std.process.run(gpa, io, .{ .argv = &.{ "objdump", "-d", "-M", "intel", "--no-show-raw-insn", "--start-address", start_addr, "--stop-address", stop_addr, path, }, }) catch |e| { try bw.print("failed to run objdump: {any}\n", .{e}); continue; }; defer gpa.free(result.stdout); defer gpa.free(result.stderr); try bw.writeAll(result.stdout); } } // --- Section collection --- pub const SectionInfo = struct { all: std.ArrayList(std.elf.Elf64_Shdr), shstrtab: ?std.elf.Elf64_Shdr = null, symtab: ?std.elf.Elf64_Shdr = null, dynsym: ?std.elf.Elf64_Shdr = null, /// String table for .symtab (resolved via sh_link) strtab: ?std.elf.Elf64_Shdr = null, /// String table for .dynsym (resolved via sh_link, typically .dynstr) dynstr: ?std.elf.Elf64_Shdr = null, pub fn deinit(self: *SectionInfo, gpa: std.mem.Allocator) void { self.all.deinit(gpa); } }; pub fn collectSections( header: std.elf.Header, reader: *std.Io.File.Reader, gpa: std.mem.Allocator, ) !SectionInfo { var info: SectionInfo = .{ .all = try .initCapacity(gpa, 8) }; var it = header.iterateSectionHeaders(reader); var idx: u32 = 0; while (try it.next()) |s| { defer idx += 1; try info.all.append(gpa, s); if (idx == header.shstrndx) { std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); info.shstrtab = s; } switch (s.sh_type) { std.elf.SHT_SYMTAB => info.symtab = s, std.elf.SHT_DYNSYM => info.dynsym = s, else => {}, } } // Resolve linked string tables via sh_link before sorting changes indices if (info.symtab) |st| if (st.sh_link < info.all.items.len) { info.strtab = info.all.items[st.sh_link]; }; if (info.dynsym) |ds| if (ds.sh_link < info.all.items.len) { info.dynstr = info.all.items[ds.sh_link]; }; std.mem.sort(std.elf.Elf64_Shdr, info.all.items, {}, struct { fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { return x.sh_addr < y.sh_addr; } }.inner); return info; } // --- Symbol collection --- pub fn collectSymbols( gpa: std.mem.Allocator, header: std.elf.Header, reader: *std.Io.File.Reader, strtab_data: ?[]const u8, dynstr_data: ?[]const u8, shstrtab_data: ?[]const u8, sections: SectionInfo, ) !std.ArrayList(SymbolRange) { var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); if (strtab_data) |data| if (sections.symtab) |st| try collectSymbolsFrom(gpa, header, reader, st, data, &syms); if (dynstr_data) |data| if (sections.dynsym) |ds| try collectSymbolsFrom(gpa, header, reader, ds, data, &syms); try collectPltSymbols(gpa, reader, sections.all.items, shstrtab_data, dynstr_data, header.is_64, header.endian, &syms); std.mem.sort(SymbolRange, syms.items, {}, struct { fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { return x.start < y.start; } }.inner); return syms; } /// Create synthetic symbols for PLT entries by parsing .rela.plt relocations. /// Each .rela.plt entry maps a GOT slot to a dynsym index; the corresponding /// PLT entry is at plt_base + (1 + i) * plt_entry_size (skipping PLT0). fn collectPltSymbols( gpa: std.mem.Allocator, reader: *std.Io.File.Reader, sections: []const std.elf.Elf64_Shdr, shstrtab_data: ?[]const u8, dynstr_data: ?[]const u8, is_64: bool, endian: std.builtin.Endian, syms: *std.ArrayList(SymbolRange), ) !void { const strtab = shstrtab_data orelse return; const dstr = dynstr_data orelse return; // Find .plt and .rela.plt by name var plt_section: ?std.elf.Elf64_Shdr = null; var rela_plt: ?std.elf.Elf64_Shdr = null; var dynsym_section: ?std.elf.Elf64_Shdr = null; for (sections) |s| { const name = std.mem.sliceTo(strtab[s.sh_name..], 0); if (std.mem.eql(u8, name, ".plt")) plt_section = s; if (std.mem.eql(u8, name, ".rela.plt")) rela_plt = s; if (s.sh_type == std.elf.SHT_DYNSYM) dynsym_section = s; } const plt = plt_section orelse return; const rela = rela_plt orelse return; const dsym = dynsym_section orelse return; const entry_size: u64 = if (plt.sh_entsize > 0) plt.sh_entsize else 16; const rela_entry_size: u64 = if (rela.sh_entsize > 0) rela.sh_entsize else @sizeOf(std.elf.Elf64_Rela); const num_entries = rela.sh_size / rela_entry_size; const sym_entry_size: u64 = if (is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); var i: u64 = 0; while (i < num_entries) : (i += 1) { // Read rela entry try reader.seekTo(rela.sh_offset + i * rela_entry_size); const rela_entry = try reader.interface.takeStruct(std.elf.Elf64_Rela, endian); // Extract symbol index from r_info (upper 32 bits on 64-bit ELF) const sym_idx = rela_entry.r_info >> 32; if (sym_idx == 0) continue; // Read the dynamic symbol to get its name const sym_offset = dsym.sh_offset + sym_idx * sym_entry_size; if (sym_offset >= dsym.sh_offset + dsym.sh_size) continue; try reader.seekTo(sym_offset); const sym = try reader.interface.takeStruct(std.elf.Elf64_Sym, endian); const base_name = std.mem.sliceTo(dstr[sym.st_name..], 0); if (base_name.len == 0) continue; // PLT entry address: skip PLT0, then entry_size per relocation const plt_addr = plt.sh_addr + (1 + i) * entry_size; const name = try std.fmt.allocPrint(gpa, "{s}@plt", .{base_name}); try syms.append(gpa, .{ .start = plt_addr, .end = plt_addr + entry_size, .name = name, .kind = std.elf.STT_FUNC, }); } } fn collectSymbolsFrom( gpa: std.mem.Allocator, header: std.elf.Header, reader: *std.Io.File.Reader, section: std.elf.Elf64_Shdr, strtab_data: []const u8, syms: *std.ArrayList(SymbolRange), ) !void { var it = iterateSymbols(header, reader, section); while (try it.next()) |s| { const name = std.mem.sliceTo(strtab_data[s.st_name..], 0); const owned = try gpa.alloc(u8, name.len); @memcpy(owned, name); try syms.append(gpa, .{ .start = s.st_value, .end = s.st_value + s.st_size, .name = owned, .kind = s.st_info & 0xf, }); } } // --- Rendering --- fn printDisassembly( gpa: std.mem.Allocator, bw: *std.Io.Writer, term: std.Io.Terminal, instrs: []cs.cs_insn, symbols: []SymbolRange, ) !void { for (instrs) |instr| { const addr = instr.address; const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct { fn inner(a: u64, sym: SymbolRange) std.math.Order { return std.math.order(a, sym.start); } }.inner); if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { try term.setColor(.blue); try bw.print("\n{x:0>16} {s}:\n", .{ addr, symbols[idx].name }); try term.setColor(.reset); } try term.setColor(.dim); try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); try term.setColor(.reset); try term.setColor(.bright_green); try bw.print("{s} ", .{instr.mnemonic}); try term.setColor(.reset); const mnemonic_len: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?); const pad: u64 = 5; for (0..(if (pad >= mnemonic_len) pad - mnemonic_len else 0)) |_| { try bw.printAsciiChar(' ', .{}); } try bw.print("{s}", .{instr.op_str}); if (try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols)) |comment| { defer gpa.free(comment); if (comment.len > 0) { try term.setColor(.blue); try bw.print(" <{s}>", .{comment}); } } try bw.print("\n", .{}); try term.setColor(.reset); } } /// Prints a hexadecimal view of the bytes. pub fn printHexdump( _: type, bw: *std.Io.Writer, term: std.Io.Terminal, bytes: []const u8, offset: u64, ) !void { const nbytes = 16; var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); while (chunks.next()) |window| { const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; try term.setColor(.dim); try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); try term.setColor(.reset); for (window, 0..) |byte, index| { try bw.print("{X:0>2} ", .{byte}); if (index == 7) try bw.writeByte(' '); } try bw.writeByte(' '); if (window.len < 16) { var missing_columns = (16 - window.len) * 3; if (window.len < 8) missing_columns += 1; try bw.splatByteAll(' ', missing_columns); } const window_bytes: []const u8 = @ptrCast(@alignCast(window)); for (window_bytes) |byte| { if (std.ascii.isPrint(byte)) { try bw.writeByte(byte); } else switch (byte) { '\n' => try bw.writeAll("␊"), '\r' => try bw.writeAll("␍"), '\t' => try bw.writeAll("␉"), else => try bw.writeByte('.'), } } try bw.writeByte('\n'); } } // --- Helpers --- fn initCapstone(header: std.elf.Header) usize { var handle: usize = undefined; const opts: struct { arch: u64, mode: u64 } = switch (header.machine) { .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 }, .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = if (header.is_64) cs.CS_MODE_64 else cs.CS_MODE_32 }, else => { std.debug.print("found machine: {any}\n", .{header.machine}); @panic("unhandled arch"); }, }; std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK); return handle; } fn allocComment( gpa: std.mem.Allocator, code: []u8, symbols: []SymbolRange, ) !?[]u8 { var iter = std.mem.splitAny(u8, code, " \t[],+-"); while (iter.next()) |s| { if (std.mem.startsWith(u8, s, "0x")) { const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { std.debug.print("{any}\n", .{e}); std.debug.dumpHex(s); break :blk 0; }; // FIXME: this algorithm isn't working to find addresses "inside" symbols if (v > 0) { const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { fn inner(a: u64, sym: SymbolRange) std.math.Order { return std.math.order(a, sym.start); } }.inner); if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { const d = v - symbols[idx].start; if (d > 0) return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); } } } } return null; } // --- Types --- pub const SymbolRange = struct { start: u64, end: u64, name: []u8, kind: u8, }; fn iterateSymbols( h: std.elf.Header, file_reader: *std.Io.File.Reader, symtab: std.elf.Elf64_Shdr, ) SymbolIterator { return .{ .elf_header = h, .file_reader = file_reader, .symtab = symtab, }; } const SymbolIterator = struct { elf_header: std.elf.Header, file_reader: *std.Io.File.Reader, symtab: std.elf.Elf64_Shdr, index: usize = 0, pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { defer it.index += 1; // TODO: handle 32-bit symbols (Elf32_Sym) — currently both branches use the same size const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); const offset = it.symtab.sh_offset + size * it.index; if (offset >= (it.symtab.sh_size + it.symtab.sh_offset)) return null; try it.file_reader.seekTo(offset); return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian); } };