diff options
Diffstat (limited to 'src/elfo-pretty.zig')
| -rw-r--r-- | src/elfo-pretty.zig | 480 |
1 files changed, 0 insertions, 480 deletions
diff --git a/src/elfo-pretty.zig b/src/elfo-pretty.zig deleted file mode 100644 index 1482670..0000000 --- a/src/elfo-pretty.zig +++ /dev/null @@ -1,480 +0,0 @@ -const std = @import("std"); -const cs = @import("capstone"); - -pub fn main(init: std.process.Init) !void { - var args = try init.minimal.args.iterateAllocator(init.gpa); - defer args.deinit(); - _ = args.next(); // skip argv[0] - - var buffer: [64]u8 = undefined; - const stderr = try init.io.lockStderr(&buffer, .escape_codes); - - // TODO: finish passing custom alloc operations here - // const mem_config: cs.cs_opt_mem = undefined; - // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK); - - var handle: usize = undefined; - defer _ = cs.cs_close(@ptrCast(&handle)); - // TODO: read the arch from the elf so we open the equivalent capstone handle - std.debug.assert(cs.cs_open(cs.CS_ARCH_X86, cs.CS_MODE_64, @ptrCast(&handle)) == cs.CS_ERR_OK); - - try printElf( - init.gpa, - init.io, - args.next() orelse "./study-samples/split", - stderr.terminal(), - handle, - .{ - // .show_unaddressable_sections = true, - // .skip_sections_content = true, - }, - ); -} - -pub fn printElf( - gpa: std.mem.Allocator, - io: std.Io, - path: []const u8, - term: std.Io.Terminal, - handle: usize, - options: struct { - show_unaddressable_sections: bool = false, - skip_sections_content: bool = false, - }, -) !void { - const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); - const bw = term.writer; - defer bw.flush() catch {}; - var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); - defer gpa.free(buffer); - - var reader = f.reader(io, buffer); - const header = try std.elf.Header.read(&reader.interface); - - const shstrtab = blk: { - var section_it = header.iterateSectionHeaders(&reader); - var section_idx: u32 = 0; - while (try section_it.next()) |s| { - defer section_idx += 1; - if (section_idx == header.shstrndx) { - std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); - break :blk s; - } - } - break :blk null; - }; - - const elf_shstrtab_slice = blk: { - if (shstrtab == null) - break :blk null; - - try reader.seekTo(shstrtab.?.sh_offset); - const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size); - break :blk slice; - }; - defer { - if (elf_shstrtab_slice != null) - gpa.free(elf_shstrtab_slice.?); - } - - const strtab = blk: { - if (elf_shstrtab_slice == null) - break :blk null; - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql( - u8, - ".strtab", - std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0), - )) - // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0) - break :blk s; - } - break :blk null; - }; - - const elf_strtab_slice = blk: { - if (strtab == null) - break :blk null; - - try reader.seekTo(strtab.?.sh_offset); - const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size); - break :blk slice; - }; - - defer { - if (elf_strtab_slice != null) - gpa.free(elf_strtab_slice.?); - } - - var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8); - { - if (elf_shstrtab_slice != null) { - var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0); - while (str_it.next()) |str| { - const owned_str = try gpa.alloc(u8, str.len); - @memcpy(owned_str, str); - try strs.append(gpa, owned_str); - } - } - } - defer { - for (strs.items) |s| { - gpa.free(s); - } - strs.deinit(gpa); - } - - var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8); - { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |section| { - try sections.append(gpa, section); - } - std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct { - pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { - return x.sh_addr < y.sh_addr; - } - }.inner); - } - defer sections.deinit(gpa); - - const symtab = blk: { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_SYMTAB) { - try bw.print("sym: {any}\n", .{s}); - break :blk s; - } - } - break :blk null; - }; - - const dynsym = blk: { - var section_it = header.iterateSectionHeaders(&reader); - while (try section_it.next()) |s| { - if (s.sh_type == std.elf.SHT_DYNSYM) { - try bw.print("sym: {any}\n", .{s}); - break :blk s; - } - } - break :blk null; - }; - try bw.print("dynsym: {any}\n", .{dynsym}); - - var symbols_index = blk: { - var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); - if (symtab != null) { - var sym_it = iterateSymbols(header, &reader, symtab.?); - while (try sym_it.next()) |s| { - const t = s.st_info & 0xf; - const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); - const owned_name = try gpa.alloc(u8, name.len); - @memcpy(owned_name, name); - try syms.append(gpa, .{ - .start = s.st_value, - .end = s.st_value + s.st_size, - .name = owned_name, - .kind = t, - }); - } - } - - // the check on elf_strtab_slice might not be necessary - if (dynsym != null and elf_strtab_slice != null) { - var sym_it = iterateSymbols(header, &reader, dynsym.?); - while (try sym_it.next()) |s| { - const t = s.st_info & 0xf; - const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); - const owned_name = try gpa.alloc(u8, name.len); - @memcpy(owned_name, name); - try syms.append(gpa, .{ - .start = s.st_value, - .end = s.st_value + s.st_size, - .name = owned_name, - .kind = t, - }); - } - } - std.mem.sort(SymbolRange, syms.items, {}, struct { - fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { - return x.start < y.start; - } - }.inner); - break :blk syms; - }; - defer { - for (symbols_index.items) |sym| { - gpa.free(sym.name); - } - symbols_index.deinit(gpa); - } - - for (symbols_index.items) |sym| { - if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) - try bw.print("{s} {x}-{x}\n", .{ sym.name, sym.start, sym.end }); - } - - for (sections.items) |section| { - if (section.sh_size > 0 and section.sh_addr > 0) { - try term.setColor(.reset); - try term.setColor(.dim); - try bw.print("\n{x}-{x} (t: {x}) -- ", .{ - section.sh_addr, - section.sh_addr + section.sh_size, - section.sh_type, - }); - try term.setColor(.bright_green); - if (elf_shstrtab_slice != null) - try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); - try bw.print("\n", .{}); - try term.setColor(.reset); - - // -- - try reader.seekTo(section.sh_offset); - - if (buffer.len < section.sh_size) { - buffer = try gpa.realloc(buffer, section.sh_size); - reader = f.reader(io, buffer); - } - const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { - switch (e) { - error.EndOfStream => { - try bw.print("failed\n", .{}); - break :blk null; - }, - error.ReadFailed => unreachable, - } - }; - // TODO: this heuristic is probably wrong - if (section_slice != null and !options.skip_sections_content) { - if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) { - const instrs: []cs.cs_insn = blk: { - var insn: [*]cs.cs_insn = undefined; - // TODO: use iter API - // https://www.capstone-engine.org/iteration.html - // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn)); - const count = cs.cs_disasm(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn)); - break :blk insn[0..count]; - }; - - try dumpInstr(gpa, bw, term, instrs, symbols_index.items); - } else { - try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); - } - } - } - } - - if (options.show_unaddressable_sections) { - for (sections.items) |section| { - if (section.sh_size > 0 and section.sh_addr == 0) { - try term.setColor(.reset); - try term.setColor(.dim); - try bw.print("{x}-{x} (t: {x}) -- ", .{ - section.sh_addr, - section.sh_addr + section.sh_size, - section.sh_type, - }); - try term.setColor(.bright_cyan); - if (elf_shstrtab_slice != null) - try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); - try bw.print("\n", .{}); - try term.setColor(.reset); - // -- - - try reader.seekTo(section.sh_offset); - - if (buffer.len < section.sh_size) { - buffer = try gpa.realloc(buffer, section.sh_size); - reader = f.reader(io, buffer); - } - const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { - switch (e) { - error.EndOfStream => { - break :blk null; - }, - error.ReadFailed => unreachable, - } - }; - if (section_slice != null and !options.skip_sections_content) { - try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); - } - } - } - } -} - -fn allocComment( - gpa: std.mem.Allocator, - code: []u8, - symbols: []SymbolRange, -) !?[]u8 { - var iter = std.mem.splitAny(u8, code, " \t[],+-"); - while (iter.next()) |s| { - if (std.mem.startsWith(u8, s, "0x")) { - // todo split at the zero char at the end of string - const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { - std.debug.print("{any}\n", .{e}); - std.debug.dumpHex(s); - break :blk 0; - }; - if (v > 0) { - const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { - fn inner(a: u64, sym: SymbolRange) std.math.Order { - return std.math.order(a, sym.start); - } - }.inner); - if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { - // if (idx > 0) - // idx -= 1; - const d = v - symbols[idx].start; - if (d > 0) - return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); - return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); - } - } - } - } - - return null; -} - -fn dumpInstr( - gpa: std.mem.Allocator, - bw: *std.Io.Writer, - term: std.Io.Terminal, - instrs: []cs.cs_insn, - symbols: []SymbolRange, -) !void { - for (instrs) |instr| { - const addr = instr.address; - const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct { - fn inner(a: u64, sym: SymbolRange) std.math.Order { - return std.math.order(a, sym.start); - } - }.inner); - - if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { - try term.setColor(.blue); - try bw.print("\n{x:0>16} {s}:\n", .{ - addr, - symbols[idx].name, - }); - try term.setColor(.reset); - } - try term.setColor(.dim); - try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); - try term.setColor(.reset); - // if(instr.detail.) - try term.setColor(.bright_green); - try bw.print("{s} ", .{instr.mnemonic}); - try term.setColor(.reset); - try bw.print("{s}", .{instr.op_str}); - const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols); - defer { - if (asm_comment != null) - gpa.free(asm_comment.?); - } - if (asm_comment != null and asm_comment.?.len > 0) { - try term.setColor(.blue); - try bw.print(" <{s}>", .{asm_comment.?}); - } - try bw.print("\n", .{}); - try term.setColor(.reset); - } -} - -/// Prints a hexadecimal view of the bytes, returning any error that occurs. -pub fn dumpHexFallible( - _: type, - bw: *std.Io.Writer, - term: std.Io.Terminal, - bytes: []const u8, - offset: u64, -) !void { - // @breakpoint(); - const nbytes = 16; - var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); - while (chunks.next()) |window| { - // 1. Print the address. - const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; - try term.setColor(.dim); - // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more. - // Also, make sure all lines are aligned by padding the address. - try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); - try term.setColor(.reset); - - // 2. Print the bytes. - for (window, 0..) |byte, index| { - try bw.print("{X:0>2} ", .{byte}); - if (index == 7) try bw.writeByte(' '); - } - try bw.writeByte(' '); - if (window.len < 16) { - var missing_columns = (16 - window.len) * 3; - if (window.len < 8) missing_columns += 1; - try bw.splatByteAll(' ', missing_columns); - } - - const window_bytes: []const u8 = @ptrCast(@alignCast(window)); - - // 3. Print the characters. - for (window_bytes) |byte| { - if (std.ascii.isPrint(byte)) { - try bw.writeByte(byte); - } else { - - // Let's print some common control codes as graphical Unicode symbols. - // We don't want to do this for all control codes because most control codes apart from - // the ones that Zig has escape sequences for are likely not very useful to print as symbols. - switch (byte) { - '\n' => try bw.writeAll("␊"), - '\r' => try bw.writeAll("␍"), - '\t' => try bw.writeAll("␉"), - else => try bw.writeByte('.'), - } - } - } - try bw.writeByte('\n'); - } -} - -const SymbolRange = struct { - start: u64, - end: u64, - name: []u8, - kind: u8, -}; - -fn iterateSymbols( - h: std.elf.Header, - file_reader: *std.Io.File.Reader, - symtab: std.elf.Elf64_Shdr, -) SymbolIterator { - return .{ - .elf_header = h, - .file_reader = file_reader, - .symtab = symtab, - }; -} - -const SymbolIterator = struct { - elf_header: std.elf.Header, - file_reader: *std.Io.File.Reader, - symtab: std.elf.Elf64_Shdr, - index: usize = 0, - - pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { - defer it.index += 1; - - const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); - const offset = it.symtab.sh_offset + size * it.index; - - if (offset >= (it.symtab.sh_size + it.symtab.sh_offset)) - return null; - - try it.file_reader.seekTo(offset); - return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian); - } -}; |
