summaryrefslogtreecommitdiff
path: root/src/elfo.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/elfo.zig')
-rw-r--r--src/elfo.zig500
1 files changed, 500 insertions, 0 deletions
diff --git a/src/elfo.zig b/src/elfo.zig
new file mode 100644
index 0000000..20f798a
--- /dev/null
+++ b/src/elfo.zig
@@ -0,0 +1,500 @@
+const std = @import("std");
+const cs = @import("capstone");
+const meta_opts = @import("meta");
+
+pub fn main(init: std.process.Init) !void {
+ var args = try init.minimal.args.iterateAllocator(init.gpa);
+ defer args.deinit();
+ _ = args.next(); // skip argv[0]
+
+ var buffer: [64]u8 = undefined;
+ const stderr = try init.io.lockStderr(&buffer, .escape_codes);
+
+ // TODO: finish passing custom alloc operations here
+ // const mem_config: cs.cs_opt_mem = undefined;
+ // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK);
+
+ if (meta_opts.gdb) @breakpoint();
+
+ try printElf(
+ init.gpa,
+ init.io,
+ args.next() orelse "./study-samples/split",
+ stderr.terminal(),
+ .{
+ .show_unaddressable_sections = true,
+ // .skip_sections_content = true,
+ },
+ );
+}
+
+pub fn printElf(
+ gpa: std.mem.Allocator,
+ io: std.Io,
+ path: []const u8,
+ term: std.Io.Terminal,
+ options: struct {
+ show_unaddressable_sections: bool = false,
+ skip_sections_content: bool = false,
+ },
+) !void {
+ const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only });
+ const bw = term.writer;
+ defer bw.flush() catch {};
+ var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100);
+ defer gpa.free(buffer);
+
+ var reader = f.reader(io, buffer);
+ const header = try std.elf.Header.read(&reader.interface);
+
+ var handle: usize = undefined;
+ defer _ = cs.cs_close(@ptrCast(&handle));
+ const opts: struct { arch: u64, mode: u64 } = switch (header.machine) {
+ .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 },
+ // The arm mode isn't working with symbols on the .text section correctly
+ .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = cs.CS_MODE_ARM },
+ else => {
+ std.debug.print("found machine: {any}\n", .{header.machine});
+ @panic("unhandled arch");
+ },
+ };
+
+ std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK);
+
+ const shstrtab = blk: {
+ var section_it = header.iterateSectionHeaders(&reader);
+ var section_idx: u32 = 0;
+ while (try section_it.next()) |s| {
+ defer section_idx += 1;
+ if (section_idx == header.shstrndx) {
+ std.debug.assert(s.sh_type == std.elf.SHT_STRTAB);
+ break :blk s;
+ }
+ }
+ break :blk null;
+ };
+
+ const elf_shstrtab_slice = blk: {
+ if (shstrtab == null)
+ break :blk null;
+
+ try reader.seekTo(shstrtab.?.sh_offset);
+ const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size);
+ break :blk slice;
+ };
+ defer {
+ if (elf_shstrtab_slice != null)
+ gpa.free(elf_shstrtab_slice.?);
+ }
+
+ const strtab = blk: {
+ if (elf_shstrtab_slice == null)
+ break :blk null;
+ var section_it = header.iterateSectionHeaders(&reader);
+ while (try section_it.next()) |s| {
+ if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql(
+ u8,
+ ".strtab",
+ std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0),
+ ))
+ // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0)
+ break :blk s;
+ }
+ break :blk null;
+ };
+
+ const elf_strtab_slice = blk: {
+ if (strtab == null)
+ break :blk null;
+
+ try reader.seekTo(strtab.?.sh_offset);
+ const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size);
+ break :blk slice;
+ };
+
+ defer {
+ if (elf_strtab_slice != null)
+ gpa.free(elf_strtab_slice.?);
+ }
+
+ var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8);
+ {
+ if (elf_shstrtab_slice != null) {
+ var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0);
+ while (str_it.next()) |str| {
+ const owned_str = try gpa.alloc(u8, str.len);
+ @memcpy(owned_str, str);
+ try strs.append(gpa, owned_str);
+ }
+ }
+ }
+ defer {
+ for (strs.items) |s| {
+ gpa.free(s);
+ }
+ strs.deinit(gpa);
+ }
+
+ var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8);
+ {
+ var section_it = header.iterateSectionHeaders(&reader);
+ while (try section_it.next()) |section| {
+ try sections.append(gpa, section);
+ }
+ std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct {
+ pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool {
+ return x.sh_addr < y.sh_addr;
+ }
+ }.inner);
+ }
+ defer sections.deinit(gpa);
+
+ const symtab = blk: {
+ var section_it = header.iterateSectionHeaders(&reader);
+ while (try section_it.next()) |s| {
+ if (s.sh_type == std.elf.SHT_SYMTAB) {
+
+ // x64: symtab: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4256, .sh_size = 1680, .sh_link = 27, .sh_info = 45, .sh_addralign = 8, .sh_entsize = 24 }
+ // arm32: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4264, .sh_size = 1856, .sh_link = 27, .sh_info = 87, .sh_addralign = 4, .sh_entsize = 16 }
+ try bw.print("symtab: {any}\n", .{s});
+ break :blk s;
+ }
+ }
+ break :blk null;
+ };
+
+ const dynsym = blk: {
+ var section_it = header.iterateSectionHeaders(&reader);
+ while (try section_it.next()) |s| {
+ if (s.sh_type == std.elf.SHT_DYNSYM) {
+ // try bw.print("sym: {any}\n", .{s});
+ break :blk s;
+ }
+ }
+ break :blk null;
+ };
+ // try bw.print("dynsym: {any}\n", .{dynsym});
+
+ var symbols_index = blk: {
+ var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8);
+ if (symtab != null) {
+ var sym_it = iterateSymbols(header, &reader, symtab.?);
+ while (try sym_it.next()) |s| {
+ const t = s.st_info & 0xf;
+ const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0);
+ const owned_name = try gpa.alloc(u8, name.len);
+ @memcpy(owned_name, name);
+ try syms.append(gpa, .{
+ .start = s.st_value,
+ .end = s.st_value + s.st_size,
+ .name = owned_name,
+ .kind = t,
+ });
+ }
+ }
+
+ // the check on elf_strtab_slice might not be necessary
+ if (dynsym != null and elf_strtab_slice != null) {
+ var sym_it = iterateSymbols(header, &reader, dynsym.?);
+ while (try sym_it.next()) |s| {
+ const t = s.st_info & 0xf;
+ const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0);
+ const owned_name = try gpa.alloc(u8, name.len);
+ @memcpy(owned_name, name);
+ try syms.append(gpa, .{
+ .start = s.st_value,
+ .end = s.st_value + s.st_size,
+ .name = owned_name,
+ .kind = t,
+ });
+ }
+ }
+ std.mem.sort(SymbolRange, syms.items, {}, struct {
+ fn inner(_: void, x: SymbolRange, y: SymbolRange) bool {
+ return x.start < y.start;
+ }
+ }.inner);
+ break :blk syms;
+ };
+ defer {
+ for (symbols_index.items) |sym| {
+ gpa.free(sym.name);
+ }
+ symbols_index.deinit(gpa);
+ }
+
+ for (symbols_index.items) |sym| {
+ if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0)
+ try bw.print("{s} {x}-{x}\n", .{ sym.name, sym.start, sym.end });
+ }
+
+ for (sections.items) |section| {
+ if (section.sh_size > 0 and section.sh_addr > 0) {
+ try term.setColor(.reset);
+ try term.setColor(.dim);
+ try bw.print("\n{x}-{x} (t: {x}) -- ", .{
+ section.sh_addr,
+ section.sh_addr + section.sh_size,
+ section.sh_type,
+ });
+ try term.setColor(.bright_green);
+ if (elf_shstrtab_slice != null)
+ try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)});
+ try bw.print("\n", .{});
+ try term.setColor(.reset);
+
+ // --
+ try reader.seekTo(section.sh_offset);
+
+ if (buffer.len < section.sh_size) {
+ buffer = try gpa.realloc(buffer, section.sh_size);
+ reader = f.reader(io, buffer);
+ }
+ const section_slice = reader.interface.take(section.sh_size) catch |e| blk: {
+ switch (e) {
+ error.EndOfStream => {
+ try bw.print("failed\n", .{});
+ break :blk null;
+ },
+ error.ReadFailed => unreachable,
+ }
+ };
+ // TODO: this heuristic is probably wrong
+ if (section_slice != null and !options.skip_sections_content) {
+ if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) {
+ const instrs: []cs.cs_insn = blk: {
+ var insn: [*]cs.cs_insn = undefined;
+ // TODO: use iter API
+ // https://www.capstone-engine.org/iteration.html
+ // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn));
+ const count = cs.cs_disasm(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn));
+ break :blk insn[0..count];
+ };
+
+ try dumpInstr(gpa, bw, term, instrs, symbols_index.items);
+ } else {
+ try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr);
+ }
+ }
+ }
+ }
+
+ if (options.show_unaddressable_sections) {
+ for (sections.items) |section| {
+ if (section.sh_size > 0 and section.sh_addr == 0) {
+ try term.setColor(.reset);
+ try term.setColor(.dim);
+ try bw.print("{x}-{x} (t: {x}) -- ", .{
+ section.sh_addr,
+ section.sh_addr + section.sh_size,
+ section.sh_type,
+ });
+ try term.setColor(.bright_cyan);
+ if (elf_shstrtab_slice != null)
+ try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)});
+ try bw.print("\n", .{});
+ try term.setColor(.reset);
+ // --
+
+ try reader.seekTo(section.sh_offset);
+
+ if (buffer.len < section.sh_size) {
+ buffer = try gpa.realloc(buffer, section.sh_size);
+ reader = f.reader(io, buffer);
+ }
+ const section_slice = reader.interface.take(section.sh_size) catch |e| blk: {
+ switch (e) {
+ error.EndOfStream => {
+ break :blk null;
+ },
+ error.ReadFailed => unreachable,
+ }
+ };
+ if (section_slice != null and !options.skip_sections_content) {
+ try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr);
+ }
+ }
+ }
+ }
+}
+
+fn allocComment(
+ gpa: std.mem.Allocator,
+ code: []u8,
+ symbols: []SymbolRange,
+) !?[]u8 {
+ var iter = std.mem.splitAny(u8, code, " \t[],+-");
+ while (iter.next()) |s| {
+ if (std.mem.startsWith(u8, s, "0x")) {
+ // todo split at the zero char at the end of string
+ const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: {
+ std.debug.print("{any}\n", .{e});
+ std.debug.dumpHex(s);
+ break :blk 0;
+ };
+ // FIXME: this algorithm isn't working to find addresses "inside" symbols
+ if (v > 0) {
+ const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct {
+ fn inner(a: u64, sym: SymbolRange) std.math.Order {
+ return std.math.order(a, sym.start);
+ }
+ }.inner);
+ if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) {
+ // if (idx > 0)
+ // idx -= 1;
+ const d = v - symbols[idx].start;
+ if (d > 0)
+ return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d });
+ return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name});
+ }
+ }
+ }
+ }
+
+ return null;
+}
+
+fn dumpInstr(
+ gpa: std.mem.Allocator,
+ bw: *std.Io.Writer,
+ term: std.Io.Terminal,
+ instrs: []cs.cs_insn,
+ symbols: []SymbolRange,
+) !void {
+ for (instrs) |instr| {
+ const addr = instr.address;
+ const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct {
+ fn inner(a: u64, sym: SymbolRange) std.math.Order {
+ return std.math.order(a, sym.start);
+ }
+ }.inner);
+
+ if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) {
+ try term.setColor(.blue);
+ try bw.print("\n{x:0>16} {s}:\n", .{
+ addr,
+ symbols[idx].name,
+ });
+ try term.setColor(.reset);
+ }
+ try term.setColor(.dim);
+ try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 });
+ try term.setColor(.reset);
+ // if(instr.detail.)
+ try term.setColor(.bright_green);
+ try bw.print("{s} ", .{instr.mnemonic});
+ try term.setColor(.reset);
+
+ const mnemonic_strlen: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?);
+ const mnemonic_pad: u64 = 10;
+ for (0..(mnemonic_pad - mnemonic_strlen)) |_| {
+ try bw.printAsciiChar(' ', .{});
+ }
+ try bw.print("{s}", .{instr.op_str});
+ const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols);
+ defer {
+ if (asm_comment != null)
+ gpa.free(asm_comment.?);
+ }
+ if (asm_comment != null and asm_comment.?.len > 0) {
+ try term.setColor(.blue);
+ try bw.print(" <{s}>", .{asm_comment.?});
+ }
+ try bw.print("\n", .{});
+ try term.setColor(.reset);
+ }
+}
+
+/// Prints a hexadecimal view of the bytes, returning any error that occurs.
+pub fn dumpHexFallible(
+ _: type,
+ bw: *std.Io.Writer,
+ term: std.Io.Terminal,
+ bytes: []const u8,
+ offset: u64,
+) !void {
+ // @breakpoint();
+ const nbytes = 16;
+ var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes);
+ while (chunks.next()) |window| {
+ // 1. Print the address.
+ const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset;
+ try term.setColor(.dim);
+ // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more.
+ // Also, make sure all lines are aligned by padding the address.
+ try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 });
+ try term.setColor(.reset);
+
+ // 2. Print the bytes.
+ for (window, 0..) |byte, index| {
+ try bw.print("{X:0>2} ", .{byte});
+ if (index == 7) try bw.writeByte(' ');
+ }
+ try bw.writeByte(' ');
+ if (window.len < 16) {
+ var missing_columns = (16 - window.len) * 3;
+ if (window.len < 8) missing_columns += 1;
+ try bw.splatByteAll(' ', missing_columns);
+ }
+
+ const window_bytes: []const u8 = @ptrCast(@alignCast(window));
+
+ // 3. Print the characters.
+ for (window_bytes) |byte| {
+ if (std.ascii.isPrint(byte)) {
+ try bw.writeByte(byte);
+ } else {
+
+ // Let's print some common control codes as graphical Unicode symbols.
+ // We don't want to do this for all control codes because most control codes apart from
+ // the ones that Zig has escape sequences for are likely not very useful to print as symbols.
+ switch (byte) {
+ '\n' => try bw.writeAll("␊"),
+ '\r' => try bw.writeAll("␍"),
+ '\t' => try bw.writeAll("␉"),
+ else => try bw.writeByte('.'),
+ }
+ }
+ }
+ try bw.writeByte('\n');
+ }
+}
+
+const SymbolRange = struct {
+ start: u64,
+ end: u64,
+ name: []u8,
+ kind: u8,
+};
+
+fn iterateSymbols(
+ h: std.elf.Header,
+ file_reader: *std.Io.File.Reader,
+ symtab: std.elf.Elf64_Shdr,
+) SymbolIterator {
+ return .{
+ .elf_header = h,
+ .file_reader = file_reader,
+ .symtab = symtab,
+ };
+}
+
+const SymbolIterator = struct {
+ elf_header: std.elf.Header,
+ file_reader: *std.Io.File.Reader,
+ symtab: std.elf.Elf64_Shdr,
+ index: usize = 0,
+
+ pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym {
+ defer it.index += 1;
+
+ const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym);
+ const offset = it.symtab.sh_offset + size * it.index;
+
+ if (offset >= (it.symtab.sh_size + it.symtab.sh_offset))
+ return null;
+
+ try it.file_reader.seekTo(offset);
+ return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian);
+ }
+};