diff options
| author | Gabriel Schneider <[email protected]> | 2026-02-19 23:21:06 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-02-19 23:21:06 -0300 |
| commit | 169ba71c6ca54be373a3bbccf17ba199bba17ca8 (patch) | |
| tree | ff7edd742f5e95f9e14e395434e124341696b7ab /src/elfo.zig | |
| parent | cea7000b4c911166faa90a6d0516ca0f12920fbb (diff) | |
| download | codenomicon-169ba71c6ca54be373a3bbccf17ba199bba17ca8.tar.gz codenomicon-169ba71c6ca54be373a3bbccf17ba199bba17ca8.zip | |
added option to run with gdb
Diffstat (limited to 'src/elfo.zig')
| -rw-r--r-- | src/elfo.zig | 500 |
1 files changed, 500 insertions, 0 deletions
diff --git a/src/elfo.zig b/src/elfo.zig new file mode 100644 index 0000000..20f798a --- /dev/null +++ b/src/elfo.zig @@ -0,0 +1,500 @@ +const std = @import("std"); +const cs = @import("capstone"); +const meta_opts = @import("meta"); + +pub fn main(init: std.process.Init) !void { + var args = try init.minimal.args.iterateAllocator(init.gpa); + defer args.deinit(); + _ = args.next(); // skip argv[0] + + var buffer: [64]u8 = undefined; + const stderr = try init.io.lockStderr(&buffer, .escape_codes); + + // TODO: finish passing custom alloc operations here + // const mem_config: cs.cs_opt_mem = undefined; + // std.debug.assert(cs.cs_option(0, cs.CS_OPT_MEM, @intFromPtr(&mem_config)) == cs.CS_ERR_OK); + + if (meta_opts.gdb) @breakpoint(); + + try printElf( + init.gpa, + init.io, + args.next() orelse "./study-samples/split", + stderr.terminal(), + .{ + .show_unaddressable_sections = true, + // .skip_sections_content = true, + }, + ); +} + +pub fn printElf( + gpa: std.mem.Allocator, + io: std.Io, + path: []const u8, + term: std.Io.Terminal, + options: struct { + show_unaddressable_sections: bool = false, + skip_sections_content: bool = false, + }, +) !void { + const f = try std.Io.Dir.cwd().openFile(io, path, .{ .mode = .read_only }); + const bw = term.writer; + defer bw.flush() catch {}; + var buffer = try gpa.alignedAlloc(u8, std.mem.Alignment.of(u64), 1024 * 100); + defer gpa.free(buffer); + + var reader = f.reader(io, buffer); + const header = try std.elf.Header.read(&reader.interface); + + var handle: usize = undefined; + defer _ = cs.cs_close(@ptrCast(&handle)); + const opts: struct { arch: u64, mode: u64 } = switch (header.machine) { + .X86_64 => .{ .arch = cs.CS_ARCH_X86, .mode = cs.CS_MODE_64 }, + // The arm mode isn't working with symbols on the .text section correctly + .ARM => .{ .arch = cs.CS_ARCH_ARM, .mode = cs.CS_MODE_ARM }, + else => { + std.debug.print("found machine: {any}\n", .{header.machine}); + @panic("unhandled arch"); + }, + }; + + std.debug.assert(cs.cs_open(@intCast(opts.arch), @intCast(opts.mode), @ptrCast(&handle)) == cs.CS_ERR_OK); + + const shstrtab = blk: { + var section_it = header.iterateSectionHeaders(&reader); + var section_idx: u32 = 0; + while (try section_it.next()) |s| { + defer section_idx += 1; + if (section_idx == header.shstrndx) { + std.debug.assert(s.sh_type == std.elf.SHT_STRTAB); + break :blk s; + } + } + break :blk null; + }; + + const elf_shstrtab_slice = blk: { + if (shstrtab == null) + break :blk null; + + try reader.seekTo(shstrtab.?.sh_offset); + const slice = try reader.interface.readAlloc(gpa, shstrtab.?.sh_size); + break :blk slice; + }; + defer { + if (elf_shstrtab_slice != null) + gpa.free(elf_shstrtab_slice.?); + } + + const strtab = blk: { + if (elf_shstrtab_slice == null) + break :blk null; + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |s| { + if (s.sh_type == std.elf.SHT_STRTAB and std.mem.eql( + u8, + ".strtab", + std.mem.sliceTo(elf_shstrtab_slice.?[s.sh_name..], 0), + )) + // if (s.sh_type == std.elf.SHT_STRTAB and s.sh_name != shstrtab.?.sh_name and s.sh_addr == 0) + break :blk s; + } + break :blk null; + }; + + const elf_strtab_slice = blk: { + if (strtab == null) + break :blk null; + + try reader.seekTo(strtab.?.sh_offset); + const slice = try reader.interface.readAlloc(gpa, strtab.?.sh_size); + break :blk slice; + }; + + defer { + if (elf_strtab_slice != null) + gpa.free(elf_strtab_slice.?); + } + + var strs: std.ArrayList([]const u8) = try .initCapacity(gpa, 8); + { + if (elf_shstrtab_slice != null) { + var str_it = std.mem.splitScalar(u8, elf_shstrtab_slice.?, 0); + while (str_it.next()) |str| { + const owned_str = try gpa.alloc(u8, str.len); + @memcpy(owned_str, str); + try strs.append(gpa, owned_str); + } + } + } + defer { + for (strs.items) |s| { + gpa.free(s); + } + strs.deinit(gpa); + } + + var sections: std.ArrayList(std.elf.Elf64_Shdr) = try .initCapacity(gpa, 8); + { + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |section| { + try sections.append(gpa, section); + } + std.mem.sort(std.elf.Elf64_Shdr, sections.items, {}, struct { + pub fn inner(_: void, x: std.elf.Elf64_Shdr, y: std.elf.Elf64_Shdr) bool { + return x.sh_addr < y.sh_addr; + } + }.inner); + } + defer sections.deinit(gpa); + + const symtab = blk: { + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |s| { + if (s.sh_type == std.elf.SHT_SYMTAB) { + + // x64: symtab: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4256, .sh_size = 1680, .sh_link = 27, .sh_info = 45, .sh_addralign = 8, .sh_entsize = 24 } + // arm32: .{ .sh_name = 1, .sh_type = 2, .sh_flags = 0, .sh_addr = 0, .sh_offset = 4264, .sh_size = 1856, .sh_link = 27, .sh_info = 87, .sh_addralign = 4, .sh_entsize = 16 } + try bw.print("symtab: {any}\n", .{s}); + break :blk s; + } + } + break :blk null; + }; + + const dynsym = blk: { + var section_it = header.iterateSectionHeaders(&reader); + while (try section_it.next()) |s| { + if (s.sh_type == std.elf.SHT_DYNSYM) { + // try bw.print("sym: {any}\n", .{s}); + break :blk s; + } + } + break :blk null; + }; + // try bw.print("dynsym: {any}\n", .{dynsym}); + + var symbols_index = blk: { + var syms: std.ArrayList(SymbolRange) = try .initCapacity(gpa, 8); + if (symtab != null) { + var sym_it = iterateSymbols(header, &reader, symtab.?); + while (try sym_it.next()) |s| { + const t = s.st_info & 0xf; + const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); + const owned_name = try gpa.alloc(u8, name.len); + @memcpy(owned_name, name); + try syms.append(gpa, .{ + .start = s.st_value, + .end = s.st_value + s.st_size, + .name = owned_name, + .kind = t, + }); + } + } + + // the check on elf_strtab_slice might not be necessary + if (dynsym != null and elf_strtab_slice != null) { + var sym_it = iterateSymbols(header, &reader, dynsym.?); + while (try sym_it.next()) |s| { + const t = s.st_info & 0xf; + const name = std.mem.sliceTo(elf_strtab_slice.?[s.st_name..], 0); + const owned_name = try gpa.alloc(u8, name.len); + @memcpy(owned_name, name); + try syms.append(gpa, .{ + .start = s.st_value, + .end = s.st_value + s.st_size, + .name = owned_name, + .kind = t, + }); + } + } + std.mem.sort(SymbolRange, syms.items, {}, struct { + fn inner(_: void, x: SymbolRange, y: SymbolRange) bool { + return x.start < y.start; + } + }.inner); + break :blk syms; + }; + defer { + for (symbols_index.items) |sym| { + gpa.free(sym.name); + } + symbols_index.deinit(gpa); + } + + for (symbols_index.items) |sym| { + if (sym.kind == std.elf.STT_FUNC and sym.name.len > 0) + try bw.print("{s} {x}-{x}\n", .{ sym.name, sym.start, sym.end }); + } + + for (sections.items) |section| { + if (section.sh_size > 0 and section.sh_addr > 0) { + try term.setColor(.reset); + try term.setColor(.dim); + try bw.print("\n{x}-{x} (t: {x}) -- ", .{ + section.sh_addr, + section.sh_addr + section.sh_size, + section.sh_type, + }); + try term.setColor(.bright_green); + if (elf_shstrtab_slice != null) + try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); + try bw.print("\n", .{}); + try term.setColor(.reset); + + // -- + try reader.seekTo(section.sh_offset); + + if (buffer.len < section.sh_size) { + buffer = try gpa.realloc(buffer, section.sh_size); + reader = f.reader(io, buffer); + } + const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { + switch (e) { + error.EndOfStream => { + try bw.print("failed\n", .{}); + break :blk null; + }, + error.ReadFailed => unreachable, + } + }; + // TODO: this heuristic is probably wrong + if (section_slice != null and !options.skip_sections_content) { + if (section.sh_type == std.elf.SHT_PROGBITS and (section.sh_flags & (std.elf.SHF_ALLOC | std.elf.SHF_EXECINSTR)) != 0) { + const instrs: []cs.cs_insn = blk: { + var insn: [*]cs.cs_insn = undefined; + // TODO: use iter API + // https://www.capstone-engine.org/iteration.html + // const count = cs.cs_disasm_iter(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, @ptrCast(&insn)); + const count = cs.cs_disasm(handle, section_slice.?.ptr, section_slice.?.len, section.sh_addr, 0, @ptrCast(&insn)); + break :blk insn[0..count]; + }; + + try dumpInstr(gpa, bw, term, instrs, symbols_index.items); + } else { + try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); + } + } + } + } + + if (options.show_unaddressable_sections) { + for (sections.items) |section| { + if (section.sh_size > 0 and section.sh_addr == 0) { + try term.setColor(.reset); + try term.setColor(.dim); + try bw.print("{x}-{x} (t: {x}) -- ", .{ + section.sh_addr, + section.sh_addr + section.sh_size, + section.sh_type, + }); + try term.setColor(.bright_cyan); + if (elf_shstrtab_slice != null) + try bw.print("{s}", .{std.mem.sliceTo(elf_shstrtab_slice.?[section.sh_name..], 0)}); + try bw.print("\n", .{}); + try term.setColor(.reset); + // -- + + try reader.seekTo(section.sh_offset); + + if (buffer.len < section.sh_size) { + buffer = try gpa.realloc(buffer, section.sh_size); + reader = f.reader(io, buffer); + } + const section_slice = reader.interface.take(section.sh_size) catch |e| blk: { + switch (e) { + error.EndOfStream => { + break :blk null; + }, + error.ReadFailed => unreachable, + } + }; + if (section_slice != null and !options.skip_sections_content) { + try dumpHexFallible(u64, bw, term, section_slice.?, section.sh_addr); + } + } + } + } +} + +fn allocComment( + gpa: std.mem.Allocator, + code: []u8, + symbols: []SymbolRange, +) !?[]u8 { + var iter = std.mem.splitAny(u8, code, " \t[],+-"); + while (iter.next()) |s| { + if (std.mem.startsWith(u8, s, "0x")) { + // todo split at the zero char at the end of string + const v = std.fmt.parseInt(u64, std.mem.sliceTo(s[2..], 0), 16) catch |e| blk: { + std.debug.print("{any}\n", .{e}); + std.debug.dumpHex(s); + break :blk 0; + }; + // FIXME: this algorithm isn't working to find addresses "inside" symbols + if (v > 0) { + const idx = std.sort.lowerBound(SymbolRange, symbols, v, struct { + fn inner(a: u64, sym: SymbolRange) std.math.Order { + return std.math.order(a, sym.start); + } + }.inner); + if (idx < symbols.len and v >= symbols[idx].start and v <= symbols[idx].end) { + // if (idx > 0) + // idx -= 1; + const d = v - symbols[idx].start; + if (d > 0) + return try std.fmt.allocPrint(gpa, "{s}+0x{x}", .{ symbols[idx].name, d }); + return try std.fmt.allocPrint(gpa, "{s}", .{symbols[idx].name}); + } + } + } + } + + return null; +} + +fn dumpInstr( + gpa: std.mem.Allocator, + bw: *std.Io.Writer, + term: std.Io.Terminal, + instrs: []cs.cs_insn, + symbols: []SymbolRange, +) !void { + for (instrs) |instr| { + const addr = instr.address; + const idx = std.sort.lowerBound(SymbolRange, symbols, addr, struct { + fn inner(a: u64, sym: SymbolRange) std.math.Order { + return std.math.order(a, sym.start); + } + }.inner); + + if (idx < symbols.len and symbols[idx].start == addr and symbols[idx].name.len > 0) { + try term.setColor(.blue); + try bw.print("\n{x:0>16} {s}:\n", .{ + addr, + symbols[idx].name, + }); + try term.setColor(.reset); + } + try term.setColor(.dim); + try bw.print("{x:0>[1]} ", .{ addr, @sizeOf(usize) * 2 }); + try term.setColor(.reset); + // if(instr.detail.) + try term.setColor(.bright_green); + try bw.print("{s} ", .{instr.mnemonic}); + try term.setColor(.reset); + + const mnemonic_strlen: u64 = @intCast(std.mem.find(u8, &instr.mnemonic, &.{0}).?); + const mnemonic_pad: u64 = 10; + for (0..(mnemonic_pad - mnemonic_strlen)) |_| { + try bw.printAsciiChar(' ', .{}); + } + try bw.print("{s}", .{instr.op_str}); + const asm_comment = try allocComment(gpa, @ptrCast(@constCast(&instr.op_str)), symbols); + defer { + if (asm_comment != null) + gpa.free(asm_comment.?); + } + if (asm_comment != null and asm_comment.?.len > 0) { + try term.setColor(.blue); + try bw.print(" <{s}>", .{asm_comment.?}); + } + try bw.print("\n", .{}); + try term.setColor(.reset); + } +} + +/// Prints a hexadecimal view of the bytes, returning any error that occurs. +pub fn dumpHexFallible( + _: type, + bw: *std.Io.Writer, + term: std.Io.Terminal, + bytes: []const u8, + offset: u64, +) !void { + // @breakpoint(); + const nbytes = 16; + var chunks = std.mem.window(u8, @ptrCast(@alignCast(bytes)), nbytes, nbytes); + while (chunks.next()) |window| { + // 1. Print the address. + const address = ((0x10 * (std.math.divCeil(usize, chunks.index orelse bytes.len, nbytes) catch unreachable)) - 0x10) + offset; + try term.setColor(.dim); + // We print the address in lowercase and the bytes in uppercase hexadecimal to distinguish them more. + // Also, make sure all lines are aligned by padding the address. + try bw.print("{x:0>[1]} ", .{ address, @sizeOf(usize) * 2 }); + try term.setColor(.reset); + + // 2. Print the bytes. + for (window, 0..) |byte, index| { + try bw.print("{X:0>2} ", .{byte}); + if (index == 7) try bw.writeByte(' '); + } + try bw.writeByte(' '); + if (window.len < 16) { + var missing_columns = (16 - window.len) * 3; + if (window.len < 8) missing_columns += 1; + try bw.splatByteAll(' ', missing_columns); + } + + const window_bytes: []const u8 = @ptrCast(@alignCast(window)); + + // 3. Print the characters. + for (window_bytes) |byte| { + if (std.ascii.isPrint(byte)) { + try bw.writeByte(byte); + } else { + + // Let's print some common control codes as graphical Unicode symbols. + // We don't want to do this for all control codes because most control codes apart from + // the ones that Zig has escape sequences for are likely not very useful to print as symbols. + switch (byte) { + '\n' => try bw.writeAll("␊"), + '\r' => try bw.writeAll("␍"), + '\t' => try bw.writeAll("␉"), + else => try bw.writeByte('.'), + } + } + } + try bw.writeByte('\n'); + } +} + +const SymbolRange = struct { + start: u64, + end: u64, + name: []u8, + kind: u8, +}; + +fn iterateSymbols( + h: std.elf.Header, + file_reader: *std.Io.File.Reader, + symtab: std.elf.Elf64_Shdr, +) SymbolIterator { + return .{ + .elf_header = h, + .file_reader = file_reader, + .symtab = symtab, + }; +} + +const SymbolIterator = struct { + elf_header: std.elf.Header, + file_reader: *std.Io.File.Reader, + symtab: std.elf.Elf64_Shdr, + index: usize = 0, + + pub fn next(it: *SymbolIterator) !?std.elf.Elf64_Sym { + defer it.index += 1; + + const size: u64 = if (it.elf_header.is_64) @sizeOf(std.elf.Elf64_Sym) else @sizeOf(std.elf.Elf64_Sym); + const offset = it.symtab.sh_offset + size * it.index; + + if (offset >= (it.symtab.sh_size + it.symtab.sh_offset)) + return null; + + try it.file_reader.seekTo(offset); + return try it.file_reader.interface.takeStruct(std.elf.Elf64_Sym, it.elf_header.endian); + } +}; |
