//! ESP32 image builder: ELF in, flashable image out, in about 200 lines of Zig. //! //! This replaces `esptool elf2image`, and it deliberately drops one of esptool's rules. esptool //! refuses two flash-mapped segments inside the same 64 KiB MMU window (bin_image.py:832-838) and //! pads the image out to the next window instead, which costs 64 KiB of zeros for a small program. //! The *device* only requires that each mapped segment satisfies //! //! (offset of its data within the image) % 64 KiB == (its load address) % 64 KiB //! //! (esp_image_format.c:903-909), and the MMU is happy to point two entries - or the same entry //! twice - at one flash page; stock ESP-IDF already aliases a page that way on every boot. Two //! segments may therefore share one window, and this builder packs them that way: a 1.4 KB program //! ships as a 1.4 KB image instead of a 66 KB one. Verified on ESP32-P4 rev v1.3 silicon. //! //! Rules the loader enforces, each learned by flashing a deliberately broken image at the board: //! * exactly two segments must land in the mapped range: on a chip with shared D/I external //! vaddr (the P4, soc.h:146-149) the loader collects them positionally and asserts //! rom_index == 2 (bootloader_utility.c:805-851); one segment aborts the boot and a third //! trips an assert inside the loop (bootloader_utility.c:842) //! * every segment length must be a multiple of 4 (esp_image_format.c:857) //! * image offset 0x20 begins an esp_app_desc_t, and min/max_efuse_blk_rev_full are read from //! it whether or not it is really a descriptor (esp_image_format.c:796-806) //! * one checksum byte sits at a 16-byte boundary, then a SHA-256 of everything before it const std = @import("std"); /// Only the ESP32-P4 is exercised by this repo; the C6 value exists because the same header field /// identifies it, and because the coprocessor on this board is one. Note that the C6 needs more /// than a different chip_id: its flash window is 0x42000000..0x43000000, its 80 MHz nibble is 0x0 /// rather than 0xF, and its bootloader really does check the app-descriptor magic word because /// SOC_MMU_PAGE_SIZE_CONFIGURABLE is set there. Building a C6 image needs those in Options. pub const Chip = enum(u16) { esp32p4 = 0x0012, esp32c6 = 0x000d, }; pub const FlashSize = enum(u4) { @"1MB" = 0, @"2MB" = 1, @"4MB" = 2, @"8MB" = 3, @"16MB" = 4, @"32MB" = 5, pub fn bytes(s: FlashSize) u32 { return @as(u32, 1) << (@as(u5, @intFromEnum(s)) + 20); } }; pub const Options = struct { chip: Chip = .esp32p4, /// Silicon revision window, major * 100 + minor. The pre-v3 P4 on this desk needs 100..199. min_rev_full: u16 = 100, max_rev_full: u16 = 199, flash_size: FlashSize = .@"16MB", flash_freq: u4 = 0xF, /// 0x02 = DIO. The bootloader reconfigures the flash from its own header anyway. flash_mode: u8 = 0x02, mmu_page: u32 = 0x10000, /// Where this image will live in flash. The device checks congruence against the ABSOLUTE /// flash address (esp_image_format.c:903-909, data_addr = flash_addr + 8), so an image built /// for a partition that is not MMU-page aligned needs different padding than one that is. flash_offset: u32 = 0x10000, /// The chip's flash-mapped vaddr window. Only the ESP32-P4's is exercised here. mapped_low: u32 = 0x40000000, mapped_high: u32 = 0x44000000, }; pub const Segment = struct { addr: u32, /// Length as written to the image, i.e. including any alignment tail. len: u32, /// Bytes of that length which exist only to line the next segment up. filler: u32, kind: Kind, pub const Kind = enum { mapped, loaded, pad }; }; pub const Layout = struct { bytes: []u8, segments: []Segment, entry: u32, payload: u32, filler: u32, overhead: u32, pub fn deinit(l: *Layout, gpa: std.mem.Allocator) void { gpa.free(l.bytes); gpa.free(l.segments); } pub const ValidateError = error{ NotTwoMappedSegments, SegmentLengthUnaligned, SegmentNotCongruent, /// Two mapped segments share a vaddr page but not a flash page: the bootloader writes one /// MMU entry per vaddr page, so the second mapping would silently replace the first. MmuEntryConflict, OverlappingSegments, TooManySegments, }; /// Re-check the finished image against the rules the device enforces, plus the MMU invariant /// the device does *not* check but silently depends on. Cheap, and it turns a board that boots /// with all its constants reading as zero into a build error. pub fn validate(l: Layout, opts: Options) ValidateError!void { if (l.segments.len > max_segments) return error.TooManySegments; var mapped: usize = 0; var off: usize = header_len; for (l.segments) |s| { if (s.len % 4 != 0) return error.SegmentLengthUnaligned; if (s.kind == .mapped) { mapped += 1; const flash_addr = opts.flash_offset + off + seg_header_len; if (flash_addr % opts.mmu_page != s.addr % opts.mmu_page) return error.SegmentNotCongruent; } off += seg_header_len + s.len; } // EXACTLY two, and the bootloader is what says so. On a chip whose D/I external vaddr ranges // are shared - the P4's are (soc.h:146-149) - ESP-IDF takes the SOC_MMU_DI_VADDR_SHARED // branch of `unpack_load_app` (bootloader_utility.c:805-851), which does not classify // segments as D or I at all: it collects them positionally into rom_addr[2] and ends with // `assert(rom_index == 2)`. One mapped segment aborts the boot with // "Assert failed in unpack_load_app, bootloader_utility.c:842 (rom_index == 2)" - measured, // by shipping one - and a third trips `assert(rom_index < 2)` inside the loop. Enforcing it // here turns a boot-time abort into a build-time error. if (mapped != 2) return error.NotTwoMappedSegments; // One MMU entry per vaddr page: any two mapped segments in the same vaddr page must come // from the same flash page, and no two segments may claim overlapping load addresses. var off_a: usize = header_len; for (l.segments, 0..) |a, i| { defer off_a += seg_header_len + a.len; if (a.kind != .mapped) continue; const a_flash = opts.flash_offset + off_a + seg_header_len; var off_b: usize = header_len; for (l.segments, 0..) |b, j| { defer off_b += seg_header_len + b.len; if (j <= i or b.kind != .mapped) continue; const b_flash = opts.flash_offset + off_b + seg_header_len; // Compare every vaddr page the two segments cover, not just the page each starts // in: a segment longer than one page maps several entries, and a conflict in any // one of them is the same silent corruption. Each segment defines an affine map // vaddr -> flash, so the flash page for a covered vaddr page is a shift away. const a_delta: i64 = @as(i64, @intCast(a_flash)) - @as(i64, a.addr); const b_delta: i64 = @as(i64, @intCast(b_flash)) - @as(i64, b.addr); const page_bytes: i64 = @intCast(opts.mmu_page); const a_first = @as(i64, a.addr) - @mod(@as(i64, a.addr), page_bytes); const a_end = @as(i64, a.addr) + @as(i64, @max(a.len, 1)); const b_first = @as(i64, b.addr) - @mod(@as(i64, b.addr), page_bytes); const b_end = @as(i64, b.addr) + @as(i64, @max(b.len, 1)); var v_page = a_first; while (v_page < a_end) : (v_page += page_bytes) { if (v_page < b_first or v_page >= b_end) continue; if (@divFloor(v_page + a_delta, page_bytes) != @divFloor(v_page + b_delta, page_bytes)) return error.MmuEntryConflict; } if (@as(i64, a.addr) < b_end and @as(i64, b.addr) < a_end) return error.OverlappingSegments; } } } }; const header_len = 24; const seg_header_len = 8; /// ESP_IMAGE_MAX_SEGMENTS (esp_app_format.h:122), enforced by esp_image_format.c:406-408. const max_segments = 16; fn isMapped(addr: u64, opts: Options) bool { return addr >= opts.mapped_low and addr < opts.mapped_high; } const Piece = struct { addr: u32, data: []const u8, /// Zero bytes appended to this segment: 0-3 for length alignment, plus however many are /// needed to make the *next* mapped segment land on a congruent offset. filler: u32 = 0, kind: Segment.Kind, }; /// Build an image from the PT_LOAD program headers of an ELF file. pub fn fromElf(gpa: std.mem.Allocator, elf_bytes: []const u8, opts: Options) !Layout { var reader: std.Io.Reader = .fixed(elf_bytes); const hdr = try std.elf.Header.read(&reader); const entry: u32 = @intCast(hdr.entry); var pieces: std.ArrayList(Piece) = .empty; defer pieces.deinit(gpa); var it = hdr.iterateProgramHeadersBuffer(elf_bytes); while (try it.next()) |ph| { if (ph.p_type != std.elf.PT_LOAD or ph.p_filesz == 0) continue; const start: usize = @intCast(ph.p_offset); const end: usize = @intCast(ph.p_offset + ph.p_filesz); const addr: u32 = @intCast(ph.p_paddr); try pieces.append(gpa, .{ .addr = addr, .data = elf_bytes[start..end], .kind = if (isMapped(addr, opts)) .mapped else .loaded, }); } if (pieces.items.len == 0) return error.NoLoadableSegments; std.mem.sort(Piece, pieces.items, {}, struct { fn lt(_: void, a: Piece, b: Piece) bool { return a.addr < b.addr; } }.lt); // Length alignment first: the loader rejects a segment whose length is not a multiple of 4. for (pieces.items) |*p| p.filler = @intCast((4 - p.data.len % 4) % 4); // Then the mapping constraint. Congruence modulo the MMU page is necessary but NOT // sufficient: the bootloader writes one MMU entry per vaddr page (mmu_hal.c:107-113, called // once per mapped segment from set_cache_and_start_app), so two mapped segments that share a // vaddr page must also share a flash page - otherwise the second write silently replaces the // first and every read through the loser resolves to the wrong flash page. A modular gap can // satisfy congruence by shifting a segment a whole page forward, which is exactly that bug. // // So anchor instead: the first mapped segment fixes delta = flash_address - load_address, and // every later mapped segment must land on the same delta. That makes the vaddr-to-flash // relationship a single translation for the whole image, which is what the MMU implements. const anchor: i64 = blk: { var off: usize = header_len; for (pieces.items) |p| { const data_off = off + seg_header_len; if (p.kind == .mapped) { const flash_addr: i64 = @as(i64, opts.flash_offset) + @as(i64, @intCast(data_off)); break :blk flash_addr - @as(i64, p.addr); } off = data_off + p.data.len + p.filler; } return error.NoMappedSegments; }; if (@mod(anchor, @as(i64, opts.mmu_page)) != 0) return error.PartitionNotPageAligned; var pass: usize = 0; while (pass <= pieces.items.len + 1) : (pass += 1) { var off: usize = header_len; var changed = false; for (pieces.items, 0..) |*p, i| { const data_off = off + seg_header_len; if (p.kind == .mapped) { const flash_addr: i64 = @as(i64, opts.flash_offset) + @as(i64, @intCast(data_off)); const want = @as(i64, p.addr) + anchor; if (flash_addr != want) { if (flash_addr > want) return error.MappedSegmentsTooClose; const gap: u32 = @intCast(want - flash_addr); if (gap % 4 != 0) return error.UnalignableGap; if (i == 0) return error.FirstSegmentMisaligned; // Grow the previous segment when it is mapped: its extra bytes live in flash // and are never read. Growing a RAM segment would make the loader copy filler // into L2MEM past the real data, so pay 8 bytes for a pad segment instead - // the loader skips those (load_addr 0 fails should_load()). const prev = &pieces.items[i - 1]; if (prev.kind == .mapped) { prev.filler += gap; } else { if (gap < seg_header_len) return error.PadTooSmall; try pieces.insert(gpa, i, .{ .addr = 0, .data = &.{}, .filler = gap - seg_header_len, .kind = .pad, }); } changed = true; break; } } off = data_off + p.data.len + p.filler; } if (!changed) break; } var out: std.ArrayList(u8) = .empty; defer out.deinit(gpa); var segs: std.ArrayList(Segment) = .empty; defer segs.deinit(gpa); if (pieces.items.len > max_segments) return error.TooManySegments; try out.appendSlice(gpa, &.{ 0xE9, @intCast(pieces.items.len), opts.flash_mode, @as(u8, opts.flash_freq) | (@as(u8, @intFromEnum(opts.flash_size)) << 4), }); try appendInt(gpa, &out, u32, entry); try out.appendSlice(gpa, &.{ 0xEE, 0, 0, 0 }); // wp_pin disabled, default drive strengths try appendInt(gpa, &out, u16, @intFromEnum(opts.chip)); try out.append(gpa, 0); // legacy min_chip_rev try appendInt(gpa, &out, u16, opts.min_rev_full); try appendInt(gpa, &out, u16, opts.max_rev_full); try out.appendSlice(gpa, &.{ 0, 0, 0, 0 }); // reserved try out.append(gpa, 1); // hash_appended std.debug.assert(out.items.len == header_len); var checksum: u8 = 0xEF; var payload: u32 = 0; var filler_total: u32 = 0; for (pieces.items) |p| { const len: u32 = @as(u32, @intCast(p.data.len)) + p.filler; try appendInt(gpa, &out, u32, p.addr); try appendInt(gpa, &out, u32, len); try out.appendSlice(gpa, p.data); try out.appendNTimes(gpa, 0, p.filler); for (p.data) |b| checksum ^= b; // filler is zero, and XOR with zero changes nothing, so it needs no accounting try segs.append(gpa, .{ .addr = p.addr, .len = len, .filler = p.filler, .kind = p.kind }); payload += @intCast(p.data.len); filler_total += p.filler; } // One checksum byte, positioned so the image length is a multiple of 16 before the digest. try out.appendNTimes(gpa, 0, (15 - out.items.len % 16) % 16); try out.append(gpa, checksum); var digest: [32]u8 = undefined; std.crypto.hash.sha2.Sha256.hash(out.items, &digest, .{}); try out.appendSlice(gpa, &digest); const bytes = try out.toOwnedSlice(gpa); return .{ .bytes = bytes, .segments = try segs.toOwnedSlice(gpa), .entry = entry, .payload = payload, .filler = filler_total, .overhead = @as(u32, @intCast(bytes.len)) - payload - filler_total, }; } fn appendInt(gpa: std.mem.Allocator, out: *std.ArrayList(u8), comptime T: type, value: T) !void { var buf: [@divExact(@bitSizeOf(T), 8)]u8 = undefined; std.mem.writeInt(T, &buf, value, .little); try out.appendSlice(gpa, &buf); } /// Parse a finished image back into its segment list. Used by the `size` step, so that reporting /// works on any image file rather than only on one the builder just produced in the same process. pub fn parse(gpa: std.mem.Allocator, bytes: []const u8, opts: Options) !Layout { if (bytes.len < header_len + 33 or bytes[0] != 0xE9) return error.NotAnEspImage; const count = bytes[1]; var segs: std.ArrayList(Segment) = .empty; errdefer segs.deinit(gpa); var off: usize = header_len; var payload: u32 = 0; for (0..count) |_| { if (off + seg_header_len > bytes.len) return error.TruncatedImage; const addr = std.mem.readInt(u32, bytes[off..][0..4], .little); const len = std.mem.readInt(u32, bytes[off + 4 ..][0..4], .little); if (off + seg_header_len + len > bytes.len) return error.TruncatedImage; try segs.append(gpa, .{ .addr = addr, .len = len, .filler = 0, // not recoverable from the image alone .kind = if (addr == 0) .pad else if (isMapped(addr, opts)) .mapped else .loaded, }); payload += len; off += seg_header_len + len; } const owned = try gpa.dupe(u8, bytes); errdefer gpa.free(owned); return .{ .bytes = owned, .segments = try segs.toOwnedSlice(gpa), .entry = std.mem.readInt(u32, bytes[4..8], .little), .payload = payload, .filler = 0, .overhead = @as(u32, @intCast(bytes.len)) - payload, }; }