From f5f8068fac59b4f16046c2022c2fc7c7e447ef4c Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 25 Aug 2026 12:40:53 -0300 Subject: zig-p4: pure-Zig ESP32-P4 toolchain build.zig generates the linker script and drives Zig's own LLD; tools/image.zig turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker. src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers; src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip; src/oracle/ diffs this HAL against ESP-IDF's on the die. --- tools/image.zig | 369 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 369 insertions(+) create mode 100644 tools/image.zig (limited to 'tools/image.zig') diff --git a/tools/image.zig b/tools/image.zig new file mode 100644 index 0000000..dea843f --- /dev/null +++ b/tools/image.zig @@ -0,0 +1,369 @@ +//! ESP32 image builder: ELF in, flashable image out, in about 200 lines of Zig. +//! +//! This replaces `esptool elf2image`, and it deliberately drops one of esptool's rules. esptool +//! refuses two flash-mapped segments inside the same 64 KiB MMU window (bin_image.py:832-838) and +//! pads the image out to the next window instead, which costs 64 KiB of zeros for a small program. +//! The *device* only requires that each mapped segment satisfies +//! +//! (offset of its data within the image) % 64 KiB == (its load address) % 64 KiB +//! +//! (esp_image_format.c:903-909), and the MMU is happy to point two entries - or the same entry +//! twice - at one flash page; stock ESP-IDF already aliases a page that way on every boot. Two +//! segments may therefore share one window, and this builder packs them that way: a 1.4 KB program +//! ships as a 1.4 KB image instead of a 66 KB one. Verified on ESP32-P4 rev v1.3 silicon. +//! +//! Rules the loader enforces, each learned by flashing a deliberately broken image at the board: +//! * exactly two segments must land in the mapped range (bootloader_utility.c:842) +//! * every segment length must be a multiple of 4 (esp_image_format.c:857) +//! * image offset 0x20 begins an esp_app_desc_t, and min/max_efuse_blk_rev_full are read from +//! it whether or not it is really a descriptor (esp_image_format.c:796-806) +//! * one checksum byte sits at a 16-byte boundary, then a SHA-256 of everything before it + +const std = @import("std"); + +/// Only the ESP32-P4 is exercised by this repo; the C6 value exists because the same header field +/// identifies it, and because the coprocessor on this board is one. Note that the C6 needs more +/// than a different chip_id: its flash window is 0x42000000..0x43000000, its 80 MHz nibble is 0x0 +/// rather than 0xF, and its bootloader really does check the app-descriptor magic word because +/// SOC_MMU_PAGE_SIZE_CONFIGURABLE is set there. Building a C6 image needs those in Options. +pub const Chip = enum(u16) { + esp32p4 = 0x0012, + esp32c6 = 0x000d, +}; + +pub const FlashSize = enum(u4) { + @"1MB" = 0, + @"2MB" = 1, + @"4MB" = 2, + @"8MB" = 3, + @"16MB" = 4, + @"32MB" = 5, + + pub fn bytes(s: FlashSize) u32 { + return @as(u32, 1) << (@as(u5, @intFromEnum(s)) + 20); + } +}; + +pub const Options = struct { + chip: Chip = .esp32p4, + /// Silicon revision window, major * 100 + minor. The pre-v3 P4 on this desk needs 100..199. + min_rev_full: u16 = 100, + max_rev_full: u16 = 199, + flash_size: FlashSize = .@"16MB", + flash_freq: u4 = 0xF, + /// 0x02 = DIO. The bootloader reconfigures the flash from its own header anyway. + flash_mode: u8 = 0x02, + mmu_page: u32 = 0x10000, + /// Where this image will live in flash. The device checks congruence against the ABSOLUTE + /// flash address (esp_image_format.c:903-909, data_addr = flash_addr + 8), so an image built + /// for a partition that is not MMU-page aligned needs different padding than one that is. + flash_offset: u32 = 0x10000, + /// The chip's flash-mapped vaddr window. Only the ESP32-P4's is exercised here. + mapped_low: u32 = 0x40000000, + mapped_high: u32 = 0x44000000, +}; + +pub const Segment = struct { + addr: u32, + /// Length as written to the image, i.e. including any alignment tail. + len: u32, + /// Bytes of that length which exist only to line the next segment up. + filler: u32, + kind: Kind, + + pub const Kind = enum { mapped, loaded, pad }; +}; + +pub const Layout = struct { + bytes: []u8, + segments: []Segment, + entry: u32, + payload: u32, + filler: u32, + overhead: u32, + + pub fn deinit(l: *Layout, gpa: std.mem.Allocator) void { + gpa.free(l.bytes); + gpa.free(l.segments); + } + + pub const ValidateError = error{ + NotTwoMappedSegments, + SegmentLengthUnaligned, + SegmentNotCongruent, + /// Two mapped segments share a vaddr page but not a flash page: the bootloader writes one + /// MMU entry per vaddr page, so the second mapping would silently replace the first. + MmuEntryConflict, + OverlappingSegments, + TooManySegments, + }; + + /// Re-check the finished image against the rules the device enforces, plus the MMU invariant + /// the device does *not* check but silently depends on. Cheap, and it turns a board that boots + /// with all its constants reading as zero into a build error. + pub fn validate(l: Layout, opts: Options) ValidateError!void { + if (l.segments.len > max_segments) return error.TooManySegments; + + var mapped: usize = 0; + var off: usize = header_len; + for (l.segments) |s| { + if (s.len % 4 != 0) return error.SegmentLengthUnaligned; + if (s.kind == .mapped) { + mapped += 1; + const flash_addr = opts.flash_offset + off + seg_header_len; + if (flash_addr % opts.mmu_page != s.addr % opts.mmu_page) + return error.SegmentNotCongruent; + } + off += seg_header_len + s.len; + } + if (mapped != 2) return error.NotTwoMappedSegments; + + // One MMU entry per vaddr page: any two mapped segments in the same vaddr page must come + // from the same flash page, and no two segments may claim overlapping load addresses. + var off_a: usize = header_len; + for (l.segments, 0..) |a, i| { + defer off_a += seg_header_len + a.len; + if (a.kind != .mapped) continue; + const a_flash = opts.flash_offset + off_a + seg_header_len; + + var off_b: usize = header_len; + for (l.segments, 0..) |b, j| { + defer off_b += seg_header_len + b.len; + if (j <= i or b.kind != .mapped) continue; + const b_flash = opts.flash_offset + off_b + seg_header_len; + // Compare every vaddr page the two segments cover, not just the page each starts + // in: a segment longer than one page maps several entries, and a conflict in any + // one of them is the same silent corruption. Each segment defines an affine map + // vaddr -> flash, so the flash page for a covered vaddr page is a shift away. + const a_delta: i64 = @as(i64, @intCast(a_flash)) - @as(i64, a.addr); + const b_delta: i64 = @as(i64, @intCast(b_flash)) - @as(i64, b.addr); + const page_bytes: i64 = @intCast(opts.mmu_page); + const a_first = @as(i64, a.addr) - @mod(@as(i64, a.addr), page_bytes); + const a_end = @as(i64, a.addr) + @as(i64, @max(a.len, 1)); + const b_first = @as(i64, b.addr) - @mod(@as(i64, b.addr), page_bytes); + const b_end = @as(i64, b.addr) + @as(i64, @max(b.len, 1)); + var v_page = a_first; + while (v_page < a_end) : (v_page += page_bytes) { + if (v_page < b_first or v_page >= b_end) continue; + if (@divFloor(v_page + a_delta, page_bytes) != @divFloor(v_page + b_delta, page_bytes)) + return error.MmuEntryConflict; + } + if (@as(i64, a.addr) < b_end and @as(i64, b.addr) < a_end) + return error.OverlappingSegments; + } + } + } +}; + +const header_len = 24; +const seg_header_len = 8; +/// ESP_IMAGE_MAX_SEGMENTS (esp_app_format.h:122), enforced by esp_image_format.c:406-408. +const max_segments = 16; + +fn isMapped(addr: u64, opts: Options) bool { + return addr >= opts.mapped_low and addr < opts.mapped_high; +} + +const Piece = struct { + addr: u32, + data: []const u8, + /// Zero bytes appended to this segment: 0-3 for length alignment, plus however many are + /// needed to make the *next* mapped segment land on a congruent offset. + filler: u32 = 0, + kind: Segment.Kind, +}; + +/// Build an image from the PT_LOAD program headers of an ELF file. +pub fn fromElf(gpa: std.mem.Allocator, elf_bytes: []const u8, opts: Options) !Layout { + var reader: std.Io.Reader = .fixed(elf_bytes); + const hdr = try std.elf.Header.read(&reader); + const entry: u32 = @intCast(hdr.entry); + + var pieces: std.ArrayList(Piece) = .empty; + defer pieces.deinit(gpa); + + var it = hdr.iterateProgramHeadersBuffer(elf_bytes); + while (try it.next()) |ph| { + if (ph.p_type != std.elf.PT_LOAD or ph.p_filesz == 0) continue; + const start: usize = @intCast(ph.p_offset); + const end: usize = @intCast(ph.p_offset + ph.p_filesz); + const addr: u32 = @intCast(ph.p_paddr); + try pieces.append(gpa, .{ + .addr = addr, + .data = elf_bytes[start..end], + .kind = if (isMapped(addr, opts)) .mapped else .loaded, + }); + } + if (pieces.items.len == 0) return error.NoLoadableSegments; + + std.mem.sort(Piece, pieces.items, {}, struct { + fn lt(_: void, a: Piece, b: Piece) bool { + return a.addr < b.addr; + } + }.lt); + + // Length alignment first: the loader rejects a segment whose length is not a multiple of 4. + for (pieces.items) |*p| p.filler = @intCast((4 - p.data.len % 4) % 4); + + // Then the mapping constraint. Congruence modulo the MMU page is necessary but NOT + // sufficient: the bootloader writes one MMU entry per vaddr page (mmu_hal.c:107-113, called + // once per mapped segment from set_cache_and_start_app), so two mapped segments that share a + // vaddr page must also share a flash page - otherwise the second write silently replaces the + // first and every read through the loser resolves to the wrong flash page. A modular gap can + // satisfy congruence by shifting a segment a whole page forward, which is exactly that bug. + // + // So anchor instead: the first mapped segment fixes delta = flash_address - load_address, and + // every later mapped segment must land on the same delta. That makes the vaddr-to-flash + // relationship a single translation for the whole image, which is what the MMU implements. + const anchor: i64 = blk: { + var off: usize = header_len; + for (pieces.items) |p| { + const data_off = off + seg_header_len; + if (p.kind == .mapped) { + const flash_addr: i64 = @as(i64, opts.flash_offset) + @as(i64, @intCast(data_off)); + break :blk flash_addr - @as(i64, p.addr); + } + off = data_off + p.data.len + p.filler; + } + return error.NoMappedSegments; + }; + if (@mod(anchor, @as(i64, opts.mmu_page)) != 0) return error.PartitionNotPageAligned; + + var pass: usize = 0; + while (pass <= pieces.items.len + 1) : (pass += 1) { + var off: usize = header_len; + var changed = false; + for (pieces.items, 0..) |*p, i| { + const data_off = off + seg_header_len; + if (p.kind == .mapped) { + const flash_addr: i64 = @as(i64, opts.flash_offset) + @as(i64, @intCast(data_off)); + const want = @as(i64, p.addr) + anchor; + if (flash_addr != want) { + if (flash_addr > want) return error.MappedSegmentsTooClose; + const gap: u32 = @intCast(want - flash_addr); + if (gap % 4 != 0) return error.UnalignableGap; + if (i == 0) return error.FirstSegmentMisaligned; + // Grow the previous segment when it is mapped: its extra bytes live in flash + // and are never read. Growing a RAM segment would make the loader copy filler + // into L2MEM past the real data, so pay 8 bytes for a pad segment instead - + // the loader skips those (load_addr 0 fails should_load()). + const prev = &pieces.items[i - 1]; + if (prev.kind == .mapped) { + prev.filler += gap; + } else { + if (gap < seg_header_len) return error.PadTooSmall; + try pieces.insert(gpa, i, .{ + .addr = 0, + .data = &.{}, + .filler = gap - seg_header_len, + .kind = .pad, + }); + } + changed = true; + break; + } + } + off = data_off + p.data.len + p.filler; + } + if (!changed) break; + } + + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + var segs: std.ArrayList(Segment) = .empty; + defer segs.deinit(gpa); + + if (pieces.items.len > max_segments) return error.TooManySegments; + try out.appendSlice(gpa, &.{ + 0xE9, + @intCast(pieces.items.len), + opts.flash_mode, + @as(u8, opts.flash_freq) | (@as(u8, @intFromEnum(opts.flash_size)) << 4), + }); + try appendInt(gpa, &out, u32, entry); + try out.appendSlice(gpa, &.{ 0xEE, 0, 0, 0 }); // wp_pin disabled, default drive strengths + try appendInt(gpa, &out, u16, @intFromEnum(opts.chip)); + try out.append(gpa, 0); // legacy min_chip_rev + try appendInt(gpa, &out, u16, opts.min_rev_full); + try appendInt(gpa, &out, u16, opts.max_rev_full); + try out.appendSlice(gpa, &.{ 0, 0, 0, 0 }); // reserved + try out.append(gpa, 1); // hash_appended + std.debug.assert(out.items.len == header_len); + + var checksum: u8 = 0xEF; + var payload: u32 = 0; + var filler_total: u32 = 0; + + for (pieces.items) |p| { + const len: u32 = @as(u32, @intCast(p.data.len)) + p.filler; + try appendInt(gpa, &out, u32, p.addr); + try appendInt(gpa, &out, u32, len); + try out.appendSlice(gpa, p.data); + try out.appendNTimes(gpa, 0, p.filler); + for (p.data) |b| checksum ^= b; + // filler is zero, and XOR with zero changes nothing, so it needs no accounting + try segs.append(gpa, .{ .addr = p.addr, .len = len, .filler = p.filler, .kind = p.kind }); + payload += @intCast(p.data.len); + filler_total += p.filler; + } + + // One checksum byte, positioned so the image length is a multiple of 16 before the digest. + try out.appendNTimes(gpa, 0, (15 - out.items.len % 16) % 16); + try out.append(gpa, checksum); + + var digest: [32]u8 = undefined; + std.crypto.hash.sha2.Sha256.hash(out.items, &digest, .{}); + try out.appendSlice(gpa, &digest); + + const bytes = try out.toOwnedSlice(gpa); + return .{ + .bytes = bytes, + .segments = try segs.toOwnedSlice(gpa), + .entry = entry, + .payload = payload, + .filler = filler_total, + .overhead = @as(u32, @intCast(bytes.len)) - payload - filler_total, + }; +} + +fn appendInt(gpa: std.mem.Allocator, out: *std.ArrayList(u8), comptime T: type, value: T) !void { + var buf: [@divExact(@bitSizeOf(T), 8)]u8 = undefined; + std.mem.writeInt(T, &buf, value, .little); + try out.appendSlice(gpa, &buf); +} + +/// Parse a finished image back into its segment list. Used by the `size` step, so that reporting +/// works on any image file rather than only on one the builder just produced in the same process. +pub fn parse(gpa: std.mem.Allocator, bytes: []const u8, opts: Options) !Layout { + if (bytes.len < header_len + 33 or bytes[0] != 0xE9) return error.NotAnEspImage; + const count = bytes[1]; + var segs: std.ArrayList(Segment) = .empty; + errdefer segs.deinit(gpa); + + var off: usize = header_len; + var payload: u32 = 0; + for (0..count) |_| { + if (off + seg_header_len > bytes.len) return error.TruncatedImage; + const addr = std.mem.readInt(u32, bytes[off..][0..4], .little); + const len = std.mem.readInt(u32, bytes[off + 4 ..][0..4], .little); + if (off + seg_header_len + len > bytes.len) return error.TruncatedImage; + try segs.append(gpa, .{ + .addr = addr, + .len = len, + .filler = 0, // not recoverable from the image alone + .kind = if (addr == 0) .pad else if (isMapped(addr, opts)) .mapped else .loaded, + }); + payload += len; + off += seg_header_len + len; + } + const owned = try gpa.dupe(u8, bytes); + errdefer gpa.free(owned); + return .{ + .bytes = owned, + .segments = try segs.toOwnedSlice(gpa), + .entry = std.mem.readInt(u32, bytes[4..8], .little), + .payload = payload, + .filler = 0, + .overhead = @as(u32, @intCast(bytes.len)) - payload, + }; +} -- cgit v1.3