diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-25 12:40:53 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-25 12:46:51 -0300 |
| commit | f5f8068fac59b4f16046c2022c2fc7c7e447ef4c (patch) | |
| tree | 2731a3ed4e51cae09e184e25778eded5fc37d1f5 /tools | |
| download | esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.tar.gz esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.zip | |
zig-p4: pure-Zig ESP32-P4 toolchain
build.zig generates the linker script and drives Zig's own LLD; tools/image.zig
turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask
ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker.
src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers;
src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip;
src/oracle/ diffs this HAL against ESP-IDF's on the die.
Diffstat (limited to 'tools')
| -rwxr-xr-x | tools/encode.sh | 55 | ||||
| -rw-r--r-- | tools/image.zig | 369 | ||||
| -rw-r--r-- | tools/image_test.zig | 387 | ||||
| -rw-r--r-- | tools/rom.zig | 262 | ||||
| -rw-r--r-- | tools/serial.zig | 200 |
5 files changed, 1273 insertions, 0 deletions
diff --git a/tools/encode.sh b/tools/encode.sh new file mode 100755 index 0000000..bbb1483 --- /dev/null +++ b/tools/encode.sh @@ -0,0 +1,55 @@ +#!/bin/sh +# Encoding oracle for the ESP32-P4's vendor ISA extensions (xespv / xesploop, "PIE"). +# +# Upstream LLVM - and therefore Zig's integrated assembler - cannot spell these mnemonics, so +# inline assembly does not close the gap: it goes through the same assembler and fails the same way. +# +# asm volatile ("esp.vld.128.ip q0, a0, 16") +# -> error: <inline asm>:1:2: unrecognized instruction mnemonic +# +# But the encodings are just words, and `.insn` places any word into the instruction stream. So +# assemble once, offline, with a toolchain that does know the mnemonics (Espressif's GAS), and bake +# the words into the source. The vendor toolchain becomes a build-time oracle, not a runtime +# dependency: nothing ships, `zig build` never invokes it, and each word sits next to the mnemonic +# it came from. Registers named by the encoding must then be pinned by hand ("{a0}" constraints), +# because LLVM cannot see through a `.insn`. +# +# Usage: tools/encode.sh 'esp.vld.128.ip q0, a0, 16' ['esp.movx.w.sar a1' ...] +# tools/encode.sh -f list.txt # one instruction per line, # comments ok +# +# Output: 0x0201223b .insn 4, 0x0201223b // esp.vld.128.ip q0, a0, 16 +# +# The -march string is load-bearing: the vendor extension is versioned and the versions do not +# share encodings. `esp.vld.128.ip q0, a0, 16` is 0x0311009f under `xespv` (defaults to 1.0) and +# 0x0201223b under `xespv2p1`. The P4 wants the latter - it is the pair in the .riscv.attributes of +# every ESP-IDF image for this chip (xesploop1p0_xespv2p1), and 0x0201223b is what Espressif's +# objdump reads back as this mnemonic out of such an image. + +set -e + +MARCH=rv32imafc_xespv2p1_xesploop1p0 + +AS=$(ls "$HOME"/.espressif/tools/riscv32-esp-elf/*/riscv32-esp-elf/bin/riscv32-esp-elf-as 2>/dev/null | head -1) +OD=$(ls "$HOME"/.espressif/tools/riscv32-esp-elf/*/riscv32-esp-elf/bin/riscv32-esp-elf-objdump 2>/dev/null | head -1) +[ -x "$AS" ] || { echo "no Espressif as found (needed only to generate encodings, never to build)" >&2; exit 1; } + +TMP=$(mktemp -d) +trap 'rm -rf "$TMP"' EXIT + +encode() { + printf '%s\n' "$1" > "$TMP/in.S" + if "$AS" -march="$MARCH" -o "$TMP/out.o" "$TMP/in.S" 2>"$TMP/err"; then + word=$("$OD" -d "$TMP/out.o" | awk '/^[[:space:]]*0:/ {print $2; exit}') + printf '0x%-10s .insn 4, 0x%s // %s\n' "$word" "$word" "$1" + else + printf '%-12s // %s -- %s\n' "REJECTED" "$1" "$(sed -n '2s/.*Error: //p' "$TMP/err")" + fi +} + +if [ "$1" = "-f" ]; then + sed 's/#.*//; s/^[[:space:]]*//; s/[[:space:]]*$//; /^$/d' "$2" | while IFS= read -r line; do + encode "$line" + done +else + for insn in "$@"; do encode "$insn"; done +fi diff --git a/tools/image.zig b/tools/image.zig new file mode 100644 index 0000000..dea843f --- /dev/null +++ b/tools/image.zig @@ -0,0 +1,369 @@ +//! ESP32 image builder: ELF in, flashable image out, in about 200 lines of Zig. +//! +//! This replaces `esptool elf2image`, and it deliberately drops one of esptool's rules. esptool +//! refuses two flash-mapped segments inside the same 64 KiB MMU window (bin_image.py:832-838) and +//! pads the image out to the next window instead, which costs 64 KiB of zeros for a small program. +//! The *device* only requires that each mapped segment satisfies +//! +//! (offset of its data within the image) % 64 KiB == (its load address) % 64 KiB +//! +//! (esp_image_format.c:903-909), and the MMU is happy to point two entries - or the same entry +//! twice - at one flash page; stock ESP-IDF already aliases a page that way on every boot. Two +//! segments may therefore share one window, and this builder packs them that way: a 1.4 KB program +//! ships as a 1.4 KB image instead of a 66 KB one. Verified on ESP32-P4 rev v1.3 silicon. +//! +//! Rules the loader enforces, each learned by flashing a deliberately broken image at the board: +//! * exactly two segments must land in the mapped range (bootloader_utility.c:842) +//! * every segment length must be a multiple of 4 (esp_image_format.c:857) +//! * image offset 0x20 begins an esp_app_desc_t, and min/max_efuse_blk_rev_full are read from +//! it whether or not it is really a descriptor (esp_image_format.c:796-806) +//! * one checksum byte sits at a 16-byte boundary, then a SHA-256 of everything before it + +const std = @import("std"); + +/// Only the ESP32-P4 is exercised by this repo; the C6 value exists because the same header field +/// identifies it, and because the coprocessor on this board is one. Note that the C6 needs more +/// than a different chip_id: its flash window is 0x42000000..0x43000000, its 80 MHz nibble is 0x0 +/// rather than 0xF, and its bootloader really does check the app-descriptor magic word because +/// SOC_MMU_PAGE_SIZE_CONFIGURABLE is set there. Building a C6 image needs those in Options. +pub const Chip = enum(u16) { + esp32p4 = 0x0012, + esp32c6 = 0x000d, +}; + +pub const FlashSize = enum(u4) { + @"1MB" = 0, + @"2MB" = 1, + @"4MB" = 2, + @"8MB" = 3, + @"16MB" = 4, + @"32MB" = 5, + + pub fn bytes(s: FlashSize) u32 { + return @as(u32, 1) << (@as(u5, @intFromEnum(s)) + 20); + } +}; + +pub const Options = struct { + chip: Chip = .esp32p4, + /// Silicon revision window, major * 100 + minor. The pre-v3 P4 on this desk needs 100..199. + min_rev_full: u16 = 100, + max_rev_full: u16 = 199, + flash_size: FlashSize = .@"16MB", + flash_freq: u4 = 0xF, + /// 0x02 = DIO. The bootloader reconfigures the flash from its own header anyway. + flash_mode: u8 = 0x02, + mmu_page: u32 = 0x10000, + /// Where this image will live in flash. The device checks congruence against the ABSOLUTE + /// flash address (esp_image_format.c:903-909, data_addr = flash_addr + 8), so an image built + /// for a partition that is not MMU-page aligned needs different padding than one that is. + flash_offset: u32 = 0x10000, + /// The chip's flash-mapped vaddr window. Only the ESP32-P4's is exercised here. + mapped_low: u32 = 0x40000000, + mapped_high: u32 = 0x44000000, +}; + +pub const Segment = struct { + addr: u32, + /// Length as written to the image, i.e. including any alignment tail. + len: u32, + /// Bytes of that length which exist only to line the next segment up. + filler: u32, + kind: Kind, + + pub const Kind = enum { mapped, loaded, pad }; +}; + +pub const Layout = struct { + bytes: []u8, + segments: []Segment, + entry: u32, + payload: u32, + filler: u32, + overhead: u32, + + pub fn deinit(l: *Layout, gpa: std.mem.Allocator) void { + gpa.free(l.bytes); + gpa.free(l.segments); + } + + pub const ValidateError = error{ + NotTwoMappedSegments, + SegmentLengthUnaligned, + SegmentNotCongruent, + /// Two mapped segments share a vaddr page but not a flash page: the bootloader writes one + /// MMU entry per vaddr page, so the second mapping would silently replace the first. + MmuEntryConflict, + OverlappingSegments, + TooManySegments, + }; + + /// Re-check the finished image against the rules the device enforces, plus the MMU invariant + /// the device does *not* check but silently depends on. Cheap, and it turns a board that boots + /// with all its constants reading as zero into a build error. + pub fn validate(l: Layout, opts: Options) ValidateError!void { + if (l.segments.len > max_segments) return error.TooManySegments; + + var mapped: usize = 0; + var off: usize = header_len; + for (l.segments) |s| { + if (s.len % 4 != 0) return error.SegmentLengthUnaligned; + if (s.kind == .mapped) { + mapped += 1; + const flash_addr = opts.flash_offset + off + seg_header_len; + if (flash_addr % opts.mmu_page != s.addr % opts.mmu_page) + return error.SegmentNotCongruent; + } + off += seg_header_len + s.len; + } + if (mapped != 2) return error.NotTwoMappedSegments; + + // One MMU entry per vaddr page: any two mapped segments in the same vaddr page must come + // from the same flash page, and no two segments may claim overlapping load addresses. + var off_a: usize = header_len; + for (l.segments, 0..) |a, i| { + defer off_a += seg_header_len + a.len; + if (a.kind != .mapped) continue; + const a_flash = opts.flash_offset + off_a + seg_header_len; + + var off_b: usize = header_len; + for (l.segments, 0..) |b, j| { + defer off_b += seg_header_len + b.len; + if (j <= i or b.kind != .mapped) continue; + const b_flash = opts.flash_offset + off_b + seg_header_len; + // Compare every vaddr page the two segments cover, not just the page each starts + // in: a segment longer than one page maps several entries, and a conflict in any + // one of them is the same silent corruption. Each segment defines an affine map + // vaddr -> flash, so the flash page for a covered vaddr page is a shift away. + const a_delta: i64 = @as(i64, @intCast(a_flash)) - @as(i64, a.addr); + const b_delta: i64 = @as(i64, @intCast(b_flash)) - @as(i64, b.addr); + const page_bytes: i64 = @intCast(opts.mmu_page); + const a_first = @as(i64, a.addr) - @mod(@as(i64, a.addr), page_bytes); + const a_end = @as(i64, a.addr) + @as(i64, @max(a.len, 1)); + const b_first = @as(i64, b.addr) - @mod(@as(i64, b.addr), page_bytes); + const b_end = @as(i64, b.addr) + @as(i64, @max(b.len, 1)); + var v_page = a_first; + while (v_page < a_end) : (v_page += page_bytes) { + if (v_page < b_first or v_page >= b_end) continue; + if (@divFloor(v_page + a_delta, page_bytes) != @divFloor(v_page + b_delta, page_bytes)) + return error.MmuEntryConflict; + } + if (@as(i64, a.addr) < b_end and @as(i64, b.addr) < a_end) + return error.OverlappingSegments; + } + } + } +}; + +const header_len = 24; +const seg_header_len = 8; +/// ESP_IMAGE_MAX_SEGMENTS (esp_app_format.h:122), enforced by esp_image_format.c:406-408. +const max_segments = 16; + +fn isMapped(addr: u64, opts: Options) bool { + return addr >= opts.mapped_low and addr < opts.mapped_high; +} + +const Piece = struct { + addr: u32, + data: []const u8, + /// Zero bytes appended to this segment: 0-3 for length alignment, plus however many are + /// needed to make the *next* mapped segment land on a congruent offset. + filler: u32 = 0, + kind: Segment.Kind, +}; + +/// Build an image from the PT_LOAD program headers of an ELF file. +pub fn fromElf(gpa: std.mem.Allocator, elf_bytes: []const u8, opts: Options) !Layout { + var reader: std.Io.Reader = .fixed(elf_bytes); + const hdr = try std.elf.Header.read(&reader); + const entry: u32 = @intCast(hdr.entry); + + var pieces: std.ArrayList(Piece) = .empty; + defer pieces.deinit(gpa); + + var it = hdr.iterateProgramHeadersBuffer(elf_bytes); + while (try it.next()) |ph| { + if (ph.p_type != std.elf.PT_LOAD or ph.p_filesz == 0) continue; + const start: usize = @intCast(ph.p_offset); + const end: usize = @intCast(ph.p_offset + ph.p_filesz); + const addr: u32 = @intCast(ph.p_paddr); + try pieces.append(gpa, .{ + .addr = addr, + .data = elf_bytes[start..end], + .kind = if (isMapped(addr, opts)) .mapped else .loaded, + }); + } + if (pieces.items.len == 0) return error.NoLoadableSegments; + + std.mem.sort(Piece, pieces.items, {}, struct { + fn lt(_: void, a: Piece, b: Piece) bool { + return a.addr < b.addr; + } + }.lt); + + // Length alignment first: the loader rejects a segment whose length is not a multiple of 4. + for (pieces.items) |*p| p.filler = @intCast((4 - p.data.len % 4) % 4); + + // Then the mapping constraint. Congruence modulo the MMU page is necessary but NOT + // sufficient: the bootloader writes one MMU entry per vaddr page (mmu_hal.c:107-113, called + // once per mapped segment from set_cache_and_start_app), so two mapped segments that share a + // vaddr page must also share a flash page - otherwise the second write silently replaces the + // first and every read through the loser resolves to the wrong flash page. A modular gap can + // satisfy congruence by shifting a segment a whole page forward, which is exactly that bug. + // + // So anchor instead: the first mapped segment fixes delta = flash_address - load_address, and + // every later mapped segment must land on the same delta. That makes the vaddr-to-flash + // relationship a single translation for the whole image, which is what the MMU implements. + const anchor: i64 = blk: { + var off: usize = header_len; + for (pieces.items) |p| { + const data_off = off + seg_header_len; + if (p.kind == .mapped) { + const flash_addr: i64 = @as(i64, opts.flash_offset) + @as(i64, @intCast(data_off)); + break :blk flash_addr - @as(i64, p.addr); + } + off = data_off + p.data.len + p.filler; + } + return error.NoMappedSegments; + }; + if (@mod(anchor, @as(i64, opts.mmu_page)) != 0) return error.PartitionNotPageAligned; + + var pass: usize = 0; + while (pass <= pieces.items.len + 1) : (pass += 1) { + var off: usize = header_len; + var changed = false; + for (pieces.items, 0..) |*p, i| { + const data_off = off + seg_header_len; + if (p.kind == .mapped) { + const flash_addr: i64 = @as(i64, opts.flash_offset) + @as(i64, @intCast(data_off)); + const want = @as(i64, p.addr) + anchor; + if (flash_addr != want) { + if (flash_addr > want) return error.MappedSegmentsTooClose; + const gap: u32 = @intCast(want - flash_addr); + if (gap % 4 != 0) return error.UnalignableGap; + if (i == 0) return error.FirstSegmentMisaligned; + // Grow the previous segment when it is mapped: its extra bytes live in flash + // and are never read. Growing a RAM segment would make the loader copy filler + // into L2MEM past the real data, so pay 8 bytes for a pad segment instead - + // the loader skips those (load_addr 0 fails should_load()). + const prev = &pieces.items[i - 1]; + if (prev.kind == .mapped) { + prev.filler += gap; + } else { + if (gap < seg_header_len) return error.PadTooSmall; + try pieces.insert(gpa, i, .{ + .addr = 0, + .data = &.{}, + .filler = gap - seg_header_len, + .kind = .pad, + }); + } + changed = true; + break; + } + } + off = data_off + p.data.len + p.filler; + } + if (!changed) break; + } + + var out: std.ArrayList(u8) = .empty; + defer out.deinit(gpa); + var segs: std.ArrayList(Segment) = .empty; + defer segs.deinit(gpa); + + if (pieces.items.len > max_segments) return error.TooManySegments; + try out.appendSlice(gpa, &.{ + 0xE9, + @intCast(pieces.items.len), + opts.flash_mode, + @as(u8, opts.flash_freq) | (@as(u8, @intFromEnum(opts.flash_size)) << 4), + }); + try appendInt(gpa, &out, u32, entry); + try out.appendSlice(gpa, &.{ 0xEE, 0, 0, 0 }); // wp_pin disabled, default drive strengths + try appendInt(gpa, &out, u16, @intFromEnum(opts.chip)); + try out.append(gpa, 0); // legacy min_chip_rev + try appendInt(gpa, &out, u16, opts.min_rev_full); + try appendInt(gpa, &out, u16, opts.max_rev_full); + try out.appendSlice(gpa, &.{ 0, 0, 0, 0 }); // reserved + try out.append(gpa, 1); // hash_appended + std.debug.assert(out.items.len == header_len); + + var checksum: u8 = 0xEF; + var payload: u32 = 0; + var filler_total: u32 = 0; + + for (pieces.items) |p| { + const len: u32 = @as(u32, @intCast(p.data.len)) + p.filler; + try appendInt(gpa, &out, u32, p.addr); + try appendInt(gpa, &out, u32, len); + try out.appendSlice(gpa, p.data); + try out.appendNTimes(gpa, 0, p.filler); + for (p.data) |b| checksum ^= b; + // filler is zero, and XOR with zero changes nothing, so it needs no accounting + try segs.append(gpa, .{ .addr = p.addr, .len = len, .filler = p.filler, .kind = p.kind }); + payload += @intCast(p.data.len); + filler_total += p.filler; + } + + // One checksum byte, positioned so the image length is a multiple of 16 before the digest. + try out.appendNTimes(gpa, 0, (15 - out.items.len % 16) % 16); + try out.append(gpa, checksum); + + var digest: [32]u8 = undefined; + std.crypto.hash.sha2.Sha256.hash(out.items, &digest, .{}); + try out.appendSlice(gpa, &digest); + + const bytes = try out.toOwnedSlice(gpa); + return .{ + .bytes = bytes, + .segments = try segs.toOwnedSlice(gpa), + .entry = entry, + .payload = payload, + .filler = filler_total, + .overhead = @as(u32, @intCast(bytes.len)) - payload - filler_total, + }; +} + +fn appendInt(gpa: std.mem.Allocator, out: *std.ArrayList(u8), comptime T: type, value: T) !void { + var buf: [@divExact(@bitSizeOf(T), 8)]u8 = undefined; + std.mem.writeInt(T, &buf, value, .little); + try out.appendSlice(gpa, &buf); +} + +/// Parse a finished image back into its segment list. Used by the `size` step, so that reporting +/// works on any image file rather than only on one the builder just produced in the same process. +pub fn parse(gpa: std.mem.Allocator, bytes: []const u8, opts: Options) !Layout { + if (bytes.len < header_len + 33 or bytes[0] != 0xE9) return error.NotAnEspImage; + const count = bytes[1]; + var segs: std.ArrayList(Segment) = .empty; + errdefer segs.deinit(gpa); + + var off: usize = header_len; + var payload: u32 = 0; + for (0..count) |_| { + if (off + seg_header_len > bytes.len) return error.TruncatedImage; + const addr = std.mem.readInt(u32, bytes[off..][0..4], .little); + const len = std.mem.readInt(u32, bytes[off + 4 ..][0..4], .little); + if (off + seg_header_len + len > bytes.len) return error.TruncatedImage; + try segs.append(gpa, .{ + .addr = addr, + .len = len, + .filler = 0, // not recoverable from the image alone + .kind = if (addr == 0) .pad else if (isMapped(addr, opts)) .mapped else .loaded, + }); + payload += len; + off += seg_header_len + len; + } + const owned = try gpa.dupe(u8, bytes); + errdefer gpa.free(owned); + return .{ + .bytes = owned, + .segments = try segs.toOwnedSlice(gpa), + .entry = std.mem.readInt(u32, bytes[4..8], .little), + .payload = payload, + .filler = 0, + .overhead = @as(u32, @intCast(bytes.len)) - payload, + }; +} diff --git a/tools/image_test.zig b/tools/image_test.zig new file mode 100644 index 0000000..9f2cbc0 --- /dev/null +++ b/tools/image_test.zig @@ -0,0 +1,387 @@ +//! Host tests for the image builder. Every case here encodes a rule the ESP32-P4 ROM bootloader +//! actually enforces, each of which was learned by flashing a deliberately broken image at the +//! board and reading the error off the serial port (see 04-report/evidence/). + +const std = @import("std"); +const image = @import("image.zig"); + +const testing = std.testing; + +/// Build a 32-bit little-endian ELF with the given PT_LOAD segments, in memory. +const Load = struct { addr: u32, len: usize }; + +fn synthElf(gpa: std.mem.Allocator, entry: u32, loads: []const Load) ![]u8 { + const ehsize = 52; + const phentsize = 32; + var out: std.ArrayList(u8) = .empty; + errdefer out.deinit(gpa); + + const phoff = ehsize; + var data_off = phoff + phentsize * loads.len; + + try out.appendSlice(gpa, &.{ 0x7F, 'E', 'L', 'F', 1, 1, 1, 0 }); // magic, 32-bit, LE, v1 + try out.appendNTimes(gpa, 0, 8); // padding + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, 2))); // ET_EXEC + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, 243))); // EM_RISCV + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, 1))); // version + try out.appendSlice(gpa, &std.mem.toBytes(entry)); + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, phoff))); + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, 0))); // shoff + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, 0))); // flags + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, ehsize))); + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, phentsize))); + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, @intCast(loads.len)))); + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, 0))); // shentsize + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, 0))); // shnum + try out.appendSlice(gpa, &std.mem.toBytes(@as(u16, 0))); // shstrndx + std.debug.assert(out.items.len == ehsize); + + for (loads) |l| { + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, 1))); // PT_LOAD + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, @intCast(data_off)))); + try out.appendSlice(gpa, &std.mem.toBytes(l.addr)); // vaddr + try out.appendSlice(gpa, &std.mem.toBytes(l.addr)); // paddr + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, @intCast(l.len)))); // filesz + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, @intCast(l.len)))); // memsz + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, 4))); // flags + try out.appendSlice(gpa, &std.mem.toBytes(@as(u32, 0x1000))); // align + data_off += l.len; + } + for (loads, 0..) |l, i| { + try out.appendNTimes(gpa, @intCast('A' + i), l.len); + } + return out.toOwnedSlice(gpa); +} + +fn segmentHeaders(bytes: []const u8) []const u8 { + return bytes[24..]; +} + +test "two mapped segments in one MMU page produce a valid, tiny image" { + const gpa = testing.allocator; + // rodata at 0x40000020 (600 B) then text at 0x40000280: 0x20+600+8 == 0x280, congruent + const elf = try synthElf(gpa, 0x40000280, &.{ + .{ .addr = 0x40000020, .len = 600 }, + .{ .addr = 0x40000280, .len = 760 }, + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + try layout.validate(.{}); + try testing.expectEqual(@as(usize, 2), layout.segments.len); // no pad segment needed + try testing.expectEqual(@as(u32, 0x40000280), layout.entry); + // 24 B header + 2*(8 B segment header) + payload + checksum pad + 32 B digest + try testing.expectEqual(@as(usize, 1440), layout.bytes.len); + try testing.expectEqual(@as(u8, 0xE9), layout.bytes[0]); + try testing.expectEqual(@as(u8, 2), layout.bytes[1]); + try testing.expectEqual(@as(u8, 1), layout.bytes[0x17]); // hash_appended +} + +test "the previous segment grows when the next mapped segment is not congruent" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40001000, &.{ + .{ .addr = 0x40000020, .len = 100 }, + .{ .addr = 0x40001000, .len = 64 }, // far away: needs padding to line up + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + try layout.validate(.{}); + // no extra segment: the first one carries filler instead of a pad segment paying a header + try testing.expectEqual(@as(usize, 2), layout.segments.len); + try testing.expect(layout.segments[0].filler > 0); + try testing.expectEqual(@as(u32, 100), layout.payload - layout.segments[1].len); +} + +test "segment lengths are padded to a multiple of four" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40000040, &.{ + .{ .addr = 0x40000020, .len = 5 }, // 5 bytes: the loader would reject this as-is + .{ .addr = 0x40000040, .len = 7 }, + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + for (layout.segments) |s| try testing.expectEqual(@as(u32, 0), s.len % 4); + try testing.expect(layout.segments[0].len >= 8); // 5 bytes rounded up, plus congruence filler +} + +test "an image with one mapped segment is rejected before it can brick a board" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40000020, &.{.{ .addr = 0x40000020, .len = 64 }}); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + try testing.expectError(error.NotTwoMappedSegments, layout.validate(.{})); +} + +test "RAM-loaded segments do not count as mapped" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40000020, &.{ + .{ .addr = 0x40000020, .len = 32 }, + .{ .addr = 0x40000060, .len = 32 }, + .{ .addr = 0x4FF00000, .len = 16 }, // L2MEM: loaded, not mapped + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + try layout.validate(.{}); + var mapped: usize = 0; + var loaded: usize = 0; + for (layout.segments) |s| switch (s.kind) { + .mapped => mapped += 1, + .loaded => loaded += 1, + .pad => {}, + }; + try testing.expectEqual(@as(usize, 2), mapped); + try testing.expectEqual(@as(usize, 1), loaded); +} + +test "the checksum byte lands on a 16-byte boundary and the digest covers everything before it" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40000280, &.{ + .{ .addr = 0x40000020, .len = 600 }, + .{ .addr = 0x40000280, .len = 760 }, + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + const checksum_off = layout.bytes.len - 33; + try testing.expectEqual(@as(usize, 15), checksum_off % 16); + + var expect: [32]u8 = undefined; + std.crypto.hash.sha2.Sha256.hash(layout.bytes[0 .. layout.bytes.len - 32], &expect, .{}); + try testing.expectEqualSlices(u8, &expect, layout.bytes[layout.bytes.len - 32 ..]); + + // and the checksum itself is the XOR of every segment byte, seeded 0xEF (filler is zero, + // so including it or not gives the same answer) + var xor: u8 = 0xEF; + var off: usize = 24; + for (layout.segments) |s| { + for (layout.bytes[off + 8 .. off + 8 + s.len]) |b| xor ^= b; + off += 8 + s.len; + } + try testing.expectEqual(xor, layout.bytes[checksum_off]); +} + +test "the header carries the revision window that keeps a pre-v3 die bootable" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40000040, &.{ + .{ .addr = 0x40000020, .len = 16 }, + .{ .addr = 0x40000040, .len = 16 }, + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{ .min_rev_full = 100, .max_rev_full = 199 }); + defer layout.deinit(gpa); + + try testing.expectEqual(@as(u16, 0x0012), std.mem.readInt(u16, layout.bytes[0x0C..0x0E], .little)); + try testing.expectEqual(@as(u16, 100), std.mem.readInt(u16, layout.bytes[0x0F..0x11], .little)); + try testing.expectEqual(@as(u16, 199), std.mem.readInt(u16, layout.bytes[0x11..0x13], .little)); +} + +test "a RAM segment never has filler copied into memory" { + const gpa = testing.allocator; + // On a P4 memory map the RAM window sorts after the flash window, so a RAM segment can never + // sit between two mapped ones - but if one ever did, growing it would copy filler into L2MEM + // past the real data. Assert the property directly rather than the mechanism. + const elf = try synthElf(gpa, 0x40001000, &.{ + .{ .addr = 0x40000020, .len = 64 }, + .{ .addr = 0x40001000, .len = 64 }, + .{ .addr = 0x4FF00000, .len = 40 }, + }); + defer gpa.free(elf); + + var layout = try image.fromElf(gpa, elf, .{}); + defer layout.deinit(gpa); + + try layout.validate(.{}); + for (layout.segments) |s| { + if (s.kind == .loaded) try testing.expectEqual(@as(u32, 0), s.filler); + } +} + +test "sweep: every rodata length either builds a device-correct image or is refused" { + // The test the second review round asked for. For a range of rodata lengths and text + // placements, either the builder refuses, or the produced BYTES satisfy an independent + // re-derivation of the device's rules - including the MMU invariant that the original solver + // violated silently. This is the only test that looks at the output rather than at an error + // name, and it is what would catch a regression in the solver, the linker-script hole, or the + // checksum layout. + const gpa = testing.allocator; + var built: usize = 0; + var refused: usize = 0; + + var len: usize = 1; + while (len <= 300) : (len += 7) { + var text: u32 = 0x40; + while (text <= 0x400) : (text += 0x20) { + const elf = try synthElf(gpa, 0x40000000 + text, &.{ + .{ .addr = 0x40000020, .len = len }, + .{ .addr = 0x40000000 + text, .len = 64 }, + }); + defer gpa.free(elf); + + var layout = image.fromElf(gpa, elf, .{}) catch { + refused += 1; + continue; + }; + defer layout.deinit(gpa); + try layout.validate(.{}); + try checkBytes(layout.bytes, 0x10000); + built += 1; + } + } + try testing.expect(built > 100); + try testing.expect(refused > 0); // the impossible layouts really are refused +} + +/// Re-derive the device's rules from a finished image, sharing no code with the builder. +fn checkBytes(bytes: []const u8, flash_offset: u32) !void { + try testing.expectEqual(@as(u8, 0xE9), bytes[0]); + const count = bytes[1]; + + var mapped: usize = 0; + var deltas: [16]i64 = undefined; + var pages: [16]u64 = undefined; + var off: usize = 24; + var xor: u8 = 0xEF; + + for (0..count) |_| { + const addr = std.mem.readInt(u32, bytes[off..][0..4], .little); + const len = std.mem.readInt(u32, bytes[off + 4 ..][0..4], .little); + try testing.expectEqual(@as(u32, 0), len % 4); // esp_image_format.c:857 + const data = bytes[off + 8 ..][0..len]; + for (data) |b| xor ^= b; + + if (addr >= 0x40000000 and addr < 0x44000000) { + const flash = @as(i64, flash_offset) + @as(i64, @intCast(off + 8)); + try testing.expectEqual(@mod(@as(i64, addr), 0x10000), @mod(flash, 0x10000)); + deltas[mapped] = flash - @as(i64, addr); + pages[mapped] = addr / 0x10000; + mapped += 1; + } + off += 8 + len; + } + try testing.expectEqual(@as(usize, 2), mapped); // bootloader_utility.c:842 + + // Two mapped segments sharing a vaddr page must share the flash page: one MMU entry each. + if (pages[0] == pages[1]) try testing.expectEqual(deltas[0], deltas[1]); + + const checksum_off = bytes.len - 33; + try testing.expectEqual(@as(usize, 15), checksum_off % 16); + try testing.expectEqual(xor, bytes[checksum_off]); + for (bytes[off..checksum_off]) |b| try testing.expectEqual(@as(u8, 0), b); + + var digest: [32]u8 = undefined; + std.crypto.hash.sha2.Sha256.hash(bytes[0 .. bytes.len - 32], &digest, .{}); + try testing.expectEqualSlices(u8, &digest, bytes[bytes.len - 32 ..]); +} + +test "parse round-trips what fromElf produced" { + // Nothing tested image.parse, and the flash and size steps both depend on it. + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x400002c0, &.{ + .{ .addr = 0x40000020, .len = 600 }, + .{ .addr = 0x400002c0, .len = 380 }, + }); + defer gpa.free(elf); + + var built = try image.fromElf(gpa, elf, .{}); + defer built.deinit(gpa); + var read_back = try image.parse(gpa, built.bytes, .{}); + defer read_back.deinit(gpa); + + try testing.expectEqual(built.entry, read_back.entry); + try testing.expectEqual(built.segments.len, read_back.segments.len); + for (built.segments, read_back.segments) |a, b| { + try testing.expectEqual(a.addr, b.addr); + try testing.expectEqual(a.len, b.len); + try testing.expectEqual(a.kind, b.kind); + } + try testing.expectEqualSlices(u8, built.bytes, read_back.bytes); +} + +test "a gap that would push a mapped segment into the next flash page is refused, not padded" { + const gpa = testing.allocator; + // The old solver shifted a whole MMU page forward to satisfy congruence modulo the page. That + // kept both segments in one *vaddr* page while putting their data in two different *flash* + // pages, so the bootloader's second MMU write replaced the first and every rodata read + // resolved to filler zeros. Found by adversarial review, reproduced by -Ddescriptor=full. + const elf = try synthElf(gpa, 0x40000140, &.{ + .{ .addr = 0x40000020, .len = 268 }, // ends at 0x12C; text at 0x140 needs data@0x140, + .{ .addr = 0x40000140, .len = 236 }, // but the next data offset is 0x134: gap 12, fine + }); + defer gpa.free(elf); + var ok_layout = try image.fromElf(gpa, elf, .{}); + defer ok_layout.deinit(gpa); + try ok_layout.validate(.{}); + try testing.expect(ok_layout.bytes.len < 1024); // no 64 KiB page jump + + // Now the pathological direction: the second mapped segment sits *before* where the first one + // already reaches, so no amount of filler can line it up. + const bad = try synthElf(gpa, 0x40000030, &.{ + .{ .addr = 0x40000020, .len = 512 }, + .{ .addr = 0x40000030, .len = 16 }, + }); + defer gpa.free(bad); + try testing.expectError(error.MappedSegmentsTooClose, image.fromElf(gpa, bad, .{})); +} + +test "validate rejects two mapped segments that would fight over one MMU entry" { + // Hand-built because the solver now refuses to produce this: both segments are congruent and + // both live in vaddr page 0x4000, but their data sits in two different flash pages, so the + // bootloader's second MMU write would replace the first. This is the shape that boots with + // every constant reading as zero. + var segs = [_]image.Segment{ + .{ .addr = 0x40000020, .len = 0x10008, .filler = 0, .kind = .mapped }, + .{ .addr = 0x40000030, .len = 16, .filler = 0, .kind = .mapped }, + }; + const layout: image.Layout = .{ + .bytes = &.{}, + .segments = &segs, + .entry = 0x40000030, + .payload = 0, + .filler = 0, + .overhead = 0, + }; + // segment 1's data lands at flash 0x10000 + (24 + 8 + 0x10008) + 8 = 0x20030: congruent + // (0x30 == 0x40000030 % 64K) but one page further along than segment 0's 0x10020. + try testing.expectError(error.MmuEntryConflict, layout.validate(.{})); +} + +test "a partition that is not MMU-page aligned is refused" { + const gpa = testing.allocator; + const elf = try synthElf(gpa, 0x40000040, &.{ + .{ .addr = 0x40000020, .len = 16 }, + .{ .addr = 0x40000040, .len = 16 }, + }); + defer gpa.free(elf); + // The device checks congruence against the absolute flash address, so an image built for + // 0x11000 needs different padding from one built for 0x10000 - and the anchor cannot be a + // whole number of pages, which means no layout satisfies the rule. + try testing.expectError(error.PartitionNotPageAligned, image.fromElf(gpa, elf, .{ .flash_offset = 0x11000 })); +} + +test "more than sixteen segments is refused, because the loader stops there" { + const gpa = testing.allocator; + var loads: [20]Load = undefined; + for (&loads, 0..) |*l, i| l.* = .{ .addr = @intCast(0x4FF00000 + i * 0x100), .len = 16 }; + loads[0] = .{ .addr = 0x40000020, .len = 16 }; + loads[1] = .{ .addr = 0x40000040, .len = 16 }; + const elf = try synthElf(gpa, 0x40000040, &loads); + defer gpa.free(elf); + try testing.expectError(error.TooManySegments, image.fromElf(gpa, elf, .{})); +} diff --git a/tools/rom.zig b/tools/rom.zig new file mode 100644 index 0000000..bbb704e --- /dev/null +++ b/tools/rom.zig @@ -0,0 +1,262 @@ +//! The Espressif ROM loader protocol, enough of it to flash a chip: SLIP framing, SYNC, flash +//! attach, and uncompressed block writes. No software stub is uploaded - the ROM can do all of +//! this by itself, and for a ~1 KB image the stub's compression and 16 KB blocks buy nothing. +//! +//! Frame format (esptool loader.py:526-534, 577): +//! request: C0 | 00 op len16 chk32 | payload | C0 +//! response: C0 | 01 op len16 val32 | data | C0 +//! with C0 -> DB DC and DB -> DB DD inside the frame. +//! +//! The ESP32 ROM loaders (unlike the ESP8266's, and unlike the software stub) append FOUR trailing +//! bytes to every response: status, reason, and two reserved bytes (esptool loader.py:653-655, +//! 676). Reading the status at data[len - 2] therefore reads a reserved byte and turns every ROM +//! error into a success - which is exactly the bug an adversarial review of this file found, after +//! driving it over a pty with a rejected FLASH_DATA block. + +const std = @import("std"); +const Port = @import("serial.zig").Port; + +pub const Cmd = enum(u8) { + flash_begin = 0x02, + flash_data = 0x03, + flash_end = 0x04, + sync = 0x08, + read_reg = 0x0A, + spi_set_params = 0x0B, + spi_attach = 0x0D, + change_baud = 0x0F, + spi_flash_md5 = 0x13, + get_security_info = 0x14, +}; + +pub const Error = error{ + SyncFailed, + CommandFailed, + ShortResponse, + Timeout, + BadFrame, +}; + +pub const Loader = struct { + port: *Port, + /// Bytes already read from the port but not yet consumed by the frame parser. Reading a byte + /// at a time costs a poll+read syscall pair each, which turned a 1.5 KB flash into a 600 ms + /// affair; refilling in bursts brings it under 60 ms. + rx: [1024]u8 = undefined, + rx_len: usize = 0, + rx_pos: usize = 0, + + /// The ROM's own block size. The stub raises this to 0x4000; we do not use the stub. + pub const block_size = 0x400; + + fn nextByte(l: *Loader, deadline_ms: i64) !?u8 { + while (l.rx_pos == l.rx_len) { + const remaining = deadline_ms - l.port.nowMs(); + if (remaining <= 0) return null; + const n = try l.port.readTimeout(&l.rx, @intCast(@min(remaining, 50))); + if (n == 0) continue; + l.rx_len = n; + l.rx_pos = 0; + } + defer l.rx_pos += 1; + return l.rx[l.rx_pos]; + } + + /// Consume input until the line has been quiet for `quiet_ms`, but never for longer than + /// twenty such windows: a board stuck in a brownout-reset loop re-prints its ROM banner + /// forever, and an unbounded version of this loop hangs the flash with no output at all. + fn drainUntilQuiet(l: *Loader, quiet_ms: i64) void { + l.rx_pos = 0; + l.rx_len = 0; + const deadline = l.port.nowMs() + 20 * quiet_ms; + while (l.port.nowMs() < deadline) { + const n = l.port.readTimeout(&l.rx, @intCast(quiet_ms)) catch return; + if (n == 0) return; + } + } + + fn frame(gpa: std.mem.Allocator, cmd: Cmd, payload: []const u8, checksum: u32) ![]u8 { + var out: std.ArrayList(u8) = .empty; + errdefer out.deinit(gpa); + var head: [8]u8 = undefined; + head[0] = 0x00; + head[1] = @intFromEnum(cmd); + std.mem.writeInt(u16, head[2..4], @intCast(payload.len), .little); + std.mem.writeInt(u32, head[4..8], checksum, .little); + + try out.append(gpa, 0xC0); + for (head) |b| try escape(gpa, &out, b); + for (payload) |b| try escape(gpa, &out, b); + try out.append(gpa, 0xC0); + return out.toOwnedSlice(gpa); + } + + fn escape(gpa: std.mem.Allocator, out: *std.ArrayList(u8), b: u8) !void { + switch (b) { + 0xC0 => try out.appendSlice(gpa, &.{ 0xDB, 0xDC }), + 0xDB => try out.appendSlice(gpa, &.{ 0xDB, 0xDD }), + else => try out.append(gpa, b), + } + } + + /// Read one SLIP frame, un-escaping as it goes. + fn readFrame(l: *Loader, gpa: std.mem.Allocator, timeout_ms: u32) ![]u8 { + var out: std.ArrayList(u8) = .empty; + errdefer out.deinit(gpa); + var started = false; + var escaping = false; + const deadline = l.port.nowMs() + @as(i64, timeout_ms); + while (try l.nextByte(deadline)) |b| { + if (!started) { + if (b == 0xC0) started = true; + continue; + } + if (escaping) { + try out.append(gpa, switch (b) { + 0xDC => 0xC0, + 0xDD => 0xDB, + else => return Error.BadFrame, + }); + escaping = false; + continue; + } + switch (b) { + 0xDB => escaping = true, + 0xC0 => { + if (out.items.len == 0) continue; // empty frame, keep looking + return out.toOwnedSlice(gpa); + }, + else => try out.append(gpa, b), + } + } + return Error.Timeout; + } + + /// Send a command and wait for its matching response. Returns the response `val` field, and + /// copies any leading response data into `out` when one is given. + pub fn command( + l: *Loader, + gpa: std.mem.Allocator, + cmd: Cmd, + payload: []const u8, + checksum: u32, + timeout_ms: u32, + out: ?[]u8, + ) !u32 { + const pkt = try frame(gpa, cmd, payload, checksum); + defer gpa.free(pkt); + try l.port.write(pkt); + + const want_data: usize = if (out) |o| o.len else 0; + var tries: usize = 0; + while (tries < 100) : (tries += 1) { + const resp = l.readFrame(gpa, timeout_ms) catch |err| return err; + defer gpa.free(resp); + // Skip anything that is not this command's reply: stale frames from a previous + // session, or the ROM's repeated SYNC echoes. + if (resp.len < 8) continue; + if (resp[0] != 0x01 or resp[1] != @intFromEnum(cmd)) continue; + + const val = std.mem.readInt(u32, resp[4..8], .little); + const data = resp[8..]; + // Status sits after the expected payload. The ROM appends four bytes (status, reason, + // two reserved) where the stub appends two; esptool tolerates either, so gate on two + // and read the status at the payload end - reading it at len-2 is what made every ROM + // error look like success. + if (data.len < want_data + 2) return Error.ShortResponse; + if (data[want_data] != 0) return Error.CommandFailed; + if (out) |o| @memcpy(o, data[0..want_data]); + return val; + } + return Error.Timeout; + } + + pub fn sync(l: *Loader, gpa: std.mem.Allocator) !void { + var payload: [36]u8 = undefined; + payload[0..4].* = .{ 0x07, 0x07, 0x12, 0x20 }; + @memset(payload[4..], 0x55); + var attempt: usize = 0; + // Short per-attempt timeout: the first SYNC after a reset usually lands before the ROM is + // listening, and waiting 200 ms for that is most of the flash time on a small image. + while (attempt < 20) : (attempt += 1) { + // A reset that did not take is the usual reason SYNC never answers, so re-run it + // periodically rather than failing the build - esptool retries the whole connect + // seven times for the same reason (loader.py:891-899). + if (attempt > 0 and attempt % 5 == 0) l.port.resetToDownload(.{}) catch {}; + if (l.command(gpa, .sync, &payload, 0, 40, null)) |_| { + // The ROM answers SYNC eight times. Swallow the echoes, but stop as soon as the + // line goes quiet instead of burning a fixed 350 ms. + l.drainUntilQuiet(15); + return; + } else |_| {} + } + return Error.SyncFailed; + } + + pub fn attachFlash(l: *Loader, gpa: std.mem.Allocator) !void { + var payload: [8]u8 = @splat(0); // default SPI pins, not legacy + _ = try l.command(gpa, .spi_attach, &payload, 0, 3000, null); + } + + pub fn setFlashParams(l: *Loader, gpa: std.mem.Allocator, total_size: u32) !void { + var payload: [24]u8 = undefined; + std.mem.writeInt(u32, payload[0..4], 0, .little); // fl_id, ignored by the ROM + std.mem.writeInt(u32, payload[4..8], total_size, .little); + std.mem.writeInt(u32, payload[8..12], 64 * 1024, .little); // block + std.mem.writeInt(u32, payload[12..16], 4 * 1024, .little); // sector + std.mem.writeInt(u32, payload[16..20], 256, .little); // page + std.mem.writeInt(u32, payload[20..24], 0xFFFF, .little); // status mask + _ = try l.command(gpa, .spi_set_params, &payload, 0, 3000, null); + } + + /// Write `data` at `offset`. The ROM erases synchronously inside FLASH_BEGIN. + pub fn writeFlash(l: *Loader, gpa: std.mem.Allocator, offset: u32, data: []const u8) !void { + const blocks: u32 = @intCast(std.math.divCeil(usize, data.len, block_size) catch unreachable); + + var begin: [20]u8 = undefined; + std.mem.writeInt(u32, begin[0..4], @intCast(data.len), .little); // erase size + std.mem.writeInt(u32, begin[4..8], blocks, .little); + std.mem.writeInt(u32, begin[8..12], block_size, .little); + std.mem.writeInt(u32, begin[12..16], offset, .little); + std.mem.writeInt(u32, begin[16..20], 0, .little); // not encrypted + const erase_timeout: u32 = @intCast(@max(@as(usize, 3000), data.len / 1024 * 30)); + _ = try l.command(gpa, .flash_begin, &begin, 0, erase_timeout, null); + + var seq: u32 = 0; + var sent: usize = 0; + var block_buf: [16 + block_size]u8 = undefined; + while (sent < data.len) : (seq += 1) { + const take = @min(block_size, data.len - sent); + const chunk = data[sent .. sent + take]; + std.mem.writeInt(u32, block_buf[0..4], block_size, .little); + std.mem.writeInt(u32, block_buf[4..8], seq, .little); + std.mem.writeInt(u32, block_buf[8..12], 0, .little); + std.mem.writeInt(u32, block_buf[12..16], 0, .little); + @memcpy(block_buf[16 .. 16 + take], chunk); + @memset(block_buf[16 + take ..], 0xFF); // ROM writes whole blocks; pad with erased value + + var checksum: u32 = 0xEF; + for (block_buf[16..]) |b| checksum ^= b; + + var attempt: usize = 0; + while (true) : (attempt += 1) { + if (l.command(gpa, .flash_data, &block_buf, checksum, 3000, null)) |_| break else |err| { + if (attempt >= 2) return err; + } + } + sent += take; + } + } + + /// MD5 over a flash range, as 32 ASCII hex characters from the ROM. Routed through + /// `command()` so the direction byte, the opcode and the status are all checked - this is the + /// only thing standing between a rejected block and a bricked image. + pub fn flashMd5(l: *Loader, gpa: std.mem.Allocator, offset: u32, len: u32, out: *[32]u8) !void { + var payload: [16]u8 = undefined; + std.mem.writeInt(u32, payload[0..4], offset, .little); + std.mem.writeInt(u32, payload[4..8], len, .little); + std.mem.writeInt(u32, payload[8..12], 0, .little); + std.mem.writeInt(u32, payload[12..16], 0, .little); + _ = try l.command(gpa, .spi_flash_md5, &payload, 0, @max(1000, len / 1024 * 8), out); + } +}; diff --git a/tools/serial.zig b/tools/serial.zig new file mode 100644 index 0000000..c13942c --- /dev/null +++ b/tools/serial.zig @@ -0,0 +1,200 @@ +//! POSIX serial port: raw mode, standard baud rates, and the DTR/RTS lines that put an Espressif +//! chip into download mode. No libc, no external tool - std.posix for termios, std.os.linux for +//! the two modem-control ioctls std does not wrap, std.Io.File for the byte traffic. +//! +//! On this board DTR drives the boot strap (GPIO35) and RTS drives CHIP_PU through a transistor +//! pair, which is why the reset sequence below needs no button press. + +const std = @import("std"); +const posix = std.posix; +const linux = std.os.linux; + +/// Any rate the hardware can divide to, not just the historical enum values: the port is +/// configured through termios2, where the baud is a plain integer. The ESP32 ROM loader +/// auto-detects the host's rate from the 0x55 pattern in SYNC, so raising this needs no +/// CHANGE_BAUDRATE handshake. +pub const Baud = enum(u32) { + b115200 = 115200, + b230400 = 230400, + b460800 = 460800, + b921600 = 921600, + b1500000 = 1500000, + b2000000 = 2000000, + + pub fn rate(b: Baud) u32 { + return @intFromEnum(b); + } +}; + +/// `struct termios2` as the Linux kernel defines it: 19 control characters, then the two integer +/// baud fields. std's `linux.termios2` uses NCCS = 32, which makes it 60 bytes instead of 44 and +/// therefore encodes TCGETS2/TCSETS2 with the wrong size field - the ioctl then fails with ENOTTY. +/// Declaring the struct here keeps the numbers right and the intent visible. +const Termios2 = extern struct { + iflag: u32, + oflag: u32, + cflag: u32, + lflag: u32, + line: u8, + cc: [19]u8, + ispeed: u32, + ospeed: u32, + + /// _IOR('T', 0x2A, struct termios2) and _IOW('T', 0x2B, struct termios2) for a 44-byte struct. + const TCGETS2: u32 = 0x802C542A; + const TCSETS2: u32 = 0x402C542B; + + /// CBAUD escape meaning "take the rate from ispeed/ospeed" (asm-generic/termbits.h). + const BOTHER: u32 = 0o010000; + const CBAUD: u32 = 0o010017; + const CS8: u32 = 0o000060; + const CREAD: u32 = 0o000200; + const CLOCAL: u32 = 0o004000; +}; + +pub const Port = struct { + file: std.Io.File, + io: std.Io, + saved: Termios2, + + // std exposes neither these ioctl numbers nor the TIOCM bits. + const TIOCEXCL = 0x540C; + const TCFLSH = 0x540B; + const TCIFLUSH = 0; + const TIOCMGET = 0x5415; + const TIOCMSET = 0x5418; + const DTR: u32 = 0x002; + const RTS: u32 = 0x004; + + pub fn open(path: []const u8, baud: Baud) !Port { + const fd = try posix.openat(posix.AT.FDCWD, path, .{ + .ACCMODE = .RDWR, + .NOCTTY = true, + .CLOEXEC = true, + }, 0); + const file: std.Io.File = .{ .handle = fd, .flags = .{ .nonblocking = false } }; + const io = std.Io.Threaded.global_single_threaded.io(); + errdefer file.close(io); + + // Configure through termios2: it is the only interface whose baud fields the kernel + // populates. `tcsetattr` uses TCSETS, whose struct has no ispeed/ospeed, so assigning + // those fields silently does nothing and leaves the port at whatever rate it had - which + // is exactly the bug that made a 1.5 KB flash take 230 ms instead of 60. + var saved: Termios2 = undefined; + if (@as(isize, @bitCast(linux.ioctl(fd, Termios2.TCGETS2, @intFromPtr(&saved)))) < 0) + return error.NotATerminal; + + var raw = saved; + // A raw byte pipe: no canonical mode, no echo, no signals, no flow control, no CR/LF + // translation, 8N1, CLOCAL so a missing carrier-detect cannot block reads, and the baud + // taken from the integer fields. + raw.iflag = 0; + raw.oflag = 0; + raw.lflag = 0; + raw.cflag = Termios2.CS8 | Termios2.CREAD | Termios2.CLOCAL | Termios2.BOTHER; + raw.ispeed = baud.rate(); + raw.ospeed = baud.rate(); + @memset(&raw.cc, 0); + raw.cc[6] = 0; // VMIN: never block for a minimum count + raw.cc[5] = 0; // VTIME: poll() owns the timeouts + if (@as(isize, @bitCast(linux.ioctl(fd, Termios2.TCSETS2, @intFromPtr(&raw)))) < 0) + return error.SetAttrFailed; + + // Claim the port exclusively, the way pyserial (and therefore esptool) does: otherwise a + // `zig build monitor` left running in another terminal splits the ROM's replies between + // two readers, and the failure looks like random SyncFailed. + if (@as(isize, @bitCast(linux.ioctl(fd, TIOCEXCL, 0))) < 0) return error.PortBusy; + // Drop anything the kernel captured at the previous line rate. + _ = linux.ioctl(fd, TCFLSH, TCIFLUSH); + + return .{ .file = file, .io = io, .saved = saved }; + } + + pub fn close(p: *Port) void { + _ = linux.ioctl(p.file.handle, Termios2.TCSETS2, @intFromPtr(&p.saved)); + p.file.close(p.io); + } + + pub fn write(p: *Port, bytes: []const u8) !void { + try p.file.writeStreamingAll(p.io, bytes); + } + + /// Read whatever is available, waiting at most `timeout_ms`. Returns 0 on timeout. + pub fn read(p: *Port, buf: []u8) !usize { + var pfd = [_]posix.pollfd{.{ .fd = p.file.handle, .events = posix.POLL.IN, .revents = 0 }}; + const ready = try posix.poll(&pfd, 0); + if (ready == 0) return 0; + return p.file.readStreaming(p.io, &.{buf}) catch |err| switch (err) { + error.EndOfStream => 0, + else => err, + }; + } + + pub fn readTimeout(p: *Port, buf: []u8, timeout_ms: i32) !usize { + var pfd = [_]posix.pollfd{.{ .fd = p.file.handle, .events = posix.POLL.IN, .revents = 0 }}; + const ready = try posix.poll(&pfd, timeout_ms); + if (ready == 0) return 0; + return p.file.readStreaming(p.io, &.{buf}) catch |err| switch (err) { + error.EndOfStream => 0, + else => err, + }; + } + + pub fn drain(p: *Port) void { + var scratch: [512]u8 = undefined; + while (true) { + const n = p.read(&scratch) catch return; + if (n == 0) return; + } + } + + fn setLines(p: *Port, dtr: bool, rts: bool) !void { + var flags: u32 = 0; + if (linux.ioctl(p.file.handle, TIOCMGET, @intFromPtr(&flags)) != 0) return error.IoctlFailed; + flags = if (dtr) flags | DTR else flags & ~DTR; + flags = if (rts) flags | RTS else flags & ~RTS; + if (linux.ioctl(p.file.handle, TIOCMSET, @intFromPtr(&flags)) != 0) return error.IoctlFailed; + } + + /// How long to hold each phase of the reset sequence. esptool uses 100/50/50 ms; the shorter + /// numbers below were measured on this board over repeated runs. Tunable because a different + /// carrier's RC network may need longer. + pub const ResetTiming = struct { + hold_reset_ms: i64 = 40, + strap_settle_ms: i64 = 25, + release_ms: i64 = 15, + }; + + /// Classic reset into the ROM download loader: hold the boot strap asserted across a reset + /// pulse. Both lines are inverted by the board's transistor pair, so an asserted RS-232 line + /// pulls its pin low. + pub fn resetToDownload(p: *Port, timing: ResetTiming) !void { + // Never leave the board held in reset because a modem-line ioctl failed halfway through. + errdefer p.setLines(false, false) catch {}; + try p.setLines(false, true); // strap released, EN low: held in reset + p.sleepMs(timing.hold_reset_ms); + try p.setLines(true, false); // strap asserted, EN high: enters the ROM loader + p.sleepMs(timing.strap_settle_ms); + try p.setLines(false, false); + p.sleepMs(timing.release_ms); + p.drain(); + } + + /// Reset and let the flashed application run. + pub fn resetToRun(p: *Port, timing: ResetTiming) !void { + errdefer p.setLines(false, false) catch {}; + try p.setLines(false, true); + p.sleepMs(timing.hold_reset_ms); + try p.setLines(false, false); + p.sleepMs(timing.release_ms); + } + + fn sleepMs(p: *Port, ms: i64) void { + std.Io.sleep(p.io, .fromMilliseconds(ms), .boot) catch {}; + } + + /// Milliseconds on a monotonic clock, for deadlines. + pub fn nowMs(p: *Port) i64 { + return std.Io.Timestamp.now(p.io, .boot).toMilliseconds(); + } +}; |
