const std = @import("std"); const cs = @import("capstone"); const meta_opts = @import("meta"); pub fn main(init: std.process.Init) !void { var args = try init.minimal.args.iterateAllocator(init.gpa); defer args.deinit(); _ = args.next(); // skip argv[0] var buffer: [64]u8 = undefined; const stderr = try init.io.lockStderr(&buffer, .escape_codes); // TODO: finish passing custom alloc operations here // const mem_config: cs.cs_opt_mem = undefined; // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK); if (meta_opts.gdb) @breakpoint(); try printElf( init.gpa, init.io, args.next() orelse "./study-samples/split", stderr.terminal(), .{ .show_unaddressable_sections = true, // .skip_sections_content = true, }, ); } pub fn printElf( gpa: std.mem.Allocator, io: std.Io, path: []const u8, term: std.Io.Terminal, options: struct { show_unaddressable_sections: bool = false, skip_sections_content: bool = false, }, ) !void { const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); const bw = term.writer; defer bw.flush() catch {}; var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); defer gpa.free(buffer); var reader = f.reader(io, buffer); const header = try std.elf.Header.read(&reader.interface); var handle: usize = undefined; defer _ = cs.cs_close(@ptrCast(&handle)); const opts: struct { arch: u64, mode: u64 } = switch (header.machine) { .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 }, // The arm mode isn't working with symbols on the .text section correctly .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = cs.CS_MODE_ARM }, else => { std.debug.print("found machine: {any}\n", .{header.machine}); @panic("unhandled arch"); }, }; std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK); const shstrtab = blk: { var section_it = header.iterateSectionHeaders(&reader); var section_idx: u32 = 0; while (try section_it.next()) |s| { defer section_idx += 1; if (section_idx == header.shstrndx) { std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); break :blk s; } } break :blk null; }; const elf_shstrtab_slice = blk: { if (shstrtab == null) break :blk null; try reader.seekTo(shstrtab.?.sh_offset); const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size); break :blk slice; }; defer { if (elf_shstrtab_slice != null) gpa.free(elf_shstrtab_slice.?); } const strtab = blk: { if (elf_shstrtab_slice == null) break :blk null; var section_it = header.iterateSectionHeaders(&reader); while (try section_it.next()) |s| { if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql( u8, ".strtab", std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0), )) // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0) break :blk s; } break :blk null; }; const elf_strtab_slice = blk: { if (strtab == null) break :blk null; try reader.seekTo(strtab.?.sh_offset); const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size); break :blk slice; }; defer { if (elf_strtab_slice != null) gpa.free(elf_strtab_slice.?); } var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8); { if (elf_shstrtab_slice != null) { var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0); while (str_it.next()) |str| { const owned_str = try gpa.alloc(u8, str.len); @memcpy(owned_str, str); try strs.append(gpa, owned_str); } } } defer { for (strs.items) |s| { gpa.free(s); } strs.deinit(gpa); } var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8); { var section_it = header.iterateSectionHeaders(&reader); while (try section_it.next()) |section| { try sections.append(gpa, section); } std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct { pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { return x.sh_addr < y.sh_addr; } }.inner); } defer sections.deinit(gpa); const symtab = blk: { var section_it = header.iterateSectionHeaders(&reader); while (try section_it.next()) |s| { if (s.sh_type == std.elf.SHT_SYMTAB) { // x64: symtab: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4256, .sh_size = 1680, .sh_link = 27, .sh_info = 45, .sh_addralign = 8, .sh_entsize = 24 } // arm32: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4264, .sh_size = 1856, .sh_link = 27, .sh_info = 87, .sh_addralign = 4, .sh_entsize = 16 } try bw.print("symtab: {any}\n", .{s}); break :blk s; } } break :blk null; }; const dynsym = blk: { var section_it = header.iterateSectionHeaders(&reader); while (try section_it.next()) |s| { if (s.sh_type == std.elf.SHT_DYNSYM) { // try bw.print("sym: {any}\n", .{s}); break :blk s; } } break :blk null; }; // try bw.print("dynsym: {any}\n", .{dynsym}); var symbols_index = blk: { var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); if (symtab != null) { var sym_it = iterateSymbols(header, &reader, symtab.?); while (try sym_it.next()) |s| { const t = s.st_info & 0xf; const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); const owned_name = try gpa.alloc(u8, name.len); @memcpy(owned_name, name); try syms.append(gpa, .{ .start = s.st_value, .end = s.st_value + s.st_size, .name = owned_name, .kind = t, }); } } // the check on elf_strtab_slice might not be necessary if (dynsym != null and elf_strtab_slice != null) { var sym_it = iterateSymbols(header, &reader, dynsym.?); while (try sym_it.next()) |s| { const t = s.st_info & 0xf; const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); const owned_name = try gpa.alloc(u8, name.len); @memcpy(owned_name, name); try syms.append(gpa, .{ .start = s.st_value, .end = s.st_value + s.st_size, .name = owned_name, .kind = t, }); } } std.mem.sort(SymbolRange, syms.items, {}, struct { fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { return x.start < y.start; } }.inner); break :blk syms; }; defer { for (symbols_index.items) |sym| { gpa.free(sym.name); } symbols_index.deinit(gpa); } for (symbols_index.items) |sym| { if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) try bw.print("{s} {x}-{x}\n", .{ sym.name, sym.start, sym.end }); } for (sections.items) |section| { if (section.sh_size > 0 and section.sh_addr > 0) { try term.setColor(.reset); try term.setColor(.dim); try bw.print("\n{x}-{x} (t: {x}) -- ", .{ section.sh_addr, section.sh_addr + section.sh_size, section.sh_type, }); try term.setColor(.bright_green); if (elf_shstrtab_slice != null) try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); try bw.print("\n", .{}); try term.setColor(.reset); // -- try reader.seekTo(section.sh_offset); if (buffer.len < section.sh_size) { buffer = try gpa.realloc(buffer, section.sh_size); reader = f.reader(io, buffer); } const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { switch (e) { error.EndOfStream => { try bw.print("failed\n", .{}); break :blk null; }, error.ReadFailed => unreachable, } }; // TODO: this heuristic is probably wrong if (section_slice != null and !options.skip_sections_content) { if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) { const instrs: []cs.cs_insn = blk: { var insn: [*]cs.cs_insn = undefined; // TODO: use iter API // https://www.capstone-engine.org/iteration.html // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn)); const count = cs.cs_disasm(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn)); break :blk insn[0..count]; }; try dumpInstr(gpa, bw, term, instrs, symbols_index.items); } else { try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); } } } } if (options.show_unaddressable_sections) { for (sections.items) |section| { if (section.sh_size > 0 and section.sh_addr == 0) { try term.setColor(.reset); try term.setColor(.dim); try bw.print("{x}-{x} (t: {x}) -- ", .{ section.sh_addr, section.sh_addr + section.sh_size, section.sh_type, }); try term.setColor(.bright_cyan); if (elf_shstrtab_slice != null) try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); try bw.print("\n", .{}); try term.setColor(.reset); // -- try reader.seekTo(section.sh_offset); if (buffer.len < section.sh_size) { buffer = try gpa.realloc(buffer, section.sh_size); reader = f.reader(io, buffer); } const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { switch (e) { error.EndOfStream => { break :blk null; }, error.ReadFailed => unreachable, } }; if (section_slice != null and !options.skip_sections_content) { try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); } } } } } fn allocComment( gpa: std.mem.Allocator, code: []u8, symbols: []SymbolRange, ) !?[]u8 { var iter = std.mem.splitAny(u8, code, " \t[],+-"); while (iter.next()) |s| { if (std.mem.startsWith(u8, s, "0x")) { // todo split at the zero char at the end of string const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { std.debug.print("{any}\n", .{e}); std.debug.dumpHex(s); break :blk 0; }; // FIXME: this algorithm isn't working to find addresses "inside" symbols if (v > 0) { const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { fn inner(a: u64, sym: SymbolRange) std.math.Order { return std.math.order(a, sym.start); } }.inner); if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { // if (idx > 0) // idx -= 1; const d = v - symbols[idx].start; if (d > 0) return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); } } } } return null; } fn dumpInstr( gpa: std.mem.Allocator, bw: *std.Io.Writer, term: std.Io.Terminal, instrs: []cs.cs_insn, symbols: []SymbolRange, ) !void { for (instrs) |instr| { const addr = instr.address; const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct { fn inner(a: u64, sym: SymbolRange) std.math.Order { return std.math.order(a, sym.start); } }.inner); if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { try term.setColor(.blue); try bw.print("\n{x:0>16} {s}:\n", .{ addr, symbols[idx].name, }); try term.setColor(.reset); } try term.setColor(.dim); try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); try term.setColor(.reset); // if(instr.detail.) try term.setColor(.bright_green); try bw.print("{s} ", .{instr.mnemonic}); try term.setColor(.reset); const mnemonic_strlen: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?); const mnemonic_pad: u64 = 10; for (0..(mnemonic_pad - mnemonic_strlen)) |_| { try bw.printAsciiChar(' ', .{}); } try bw.print("{s}", .{instr.op_str}); const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols); defer { if (asm_comment != null) gpa.free(asm_comment.?); } if (asm_comment != null and asm_comment.?.len > 0) { try term.setColor(.blue); try bw.print(" <{s}>", .{asm_comment.?}); } try bw.print("\n", .{}); try term.setColor(.reset); } } /// Prints a hexadecimal view of the bytes, returning any error that occurs. pub fn dumpHexFallible( _: type, bw: *std.Io.Writer, term: std.Io.Terminal, bytes: []const u8, offset: u64, ) !void { // @breakpoint(); const nbytes = 16; var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); while (chunks.next()) |window| { // 1. Print the address. const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; try term.setColor(.dim); // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more. // Also, make sure all lines are aligned by padding the address. try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); try term.setColor(.reset); // 2. Print the bytes. for (window, 0..) |byte, index| { try bw.print("{X:0>2} ", .{byte}); if (index == 7) try bw.writeByte(' '); } try bw.writeByte(' '); if (window.len < 16) { var missing_columns = (16 - window.len) * 3; if (window.len < 8) missing_columns += 1; try bw.splatByteAll(' ', missing_columns); } const window_bytes: []const u8 = @ptrCast(@alignCast(window)); // 3. Print the characters. for (window_bytes) |byte| { if (std.ascii.isPrint(byte)) { try bw.writeByte(byte); } else { // Let's print some common control codes as graphical Unicode symbols. // We don't want to do this for all control codes because most control codes apart from // the ones that Zig has escape sequences for are likely not very useful to print as symbols. switch (byte) { '\n' => try bw.writeAll("␊"), '\r' => try bw.writeAll("␍"), '\t' => try bw.writeAll("␉"), else => try bw.writeByte('.'), } } } try bw.writeByte('\n'); } } const SymbolRange = struct { start: u64, end: u64, name: []u8, kind: u8, }; fn iterateSymbols( h: std.elf.Header, file_reader: *std.Io.File.Reader, symtab: std.elf.Elf64_Shdr, ) SymbolIterator { return .{ .elf_header = h, .file_reader = file_reader, .symtab = symtab, }; } const SymbolIterator = struct { elf_header: std.elf.Header, file_reader: *std.Io.File.Reader, symtab: std.elf.Elf64_Shdr, index: usize = 0, pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { defer it.index += 1; const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); const offset = it.symtab.sh_offset + size * it.index; if (offset >= (it.symtab.sh_size + it.symtab.sh_offset)) return null; try it.file_reader.seekTo(offset); return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian); } };