diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-25 12:40:53 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-25 12:46:51 -0300 |
| commit | f5f8068fac59b4f16046c2022c2fc7c7e447ef4c (patch) | |
| tree | 2731a3ed4e51cae09e184e25778eded5fc37d1f5 /src | |
| download | esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.tar.gz esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.zip | |
zig-p4: pure-Zig ESP32-P4 toolchain
build.zig generates the linker script and drives Zig's own LLD; tools/image.zig
turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask
ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker.
src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers;
src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip;
src/oracle/ diffs this HAL against ESP-IDF's on the die.
Diffstat (limited to 'src')
55 files changed, 28137 insertions, 0 deletions
diff --git a/src/appdesc.zig b/src/appdesc.zig new file mode 100644 index 0000000..bec2915 --- /dev/null +++ b/src/appdesc.zig @@ -0,0 +1,63 @@ +//! esp_app_desc_t, the 256-byte block the second-stage bootloader reads from image offset 0x20. +//! +//! Two facts about it were established by flashing deliberately broken images at this board: +//! * the magic word is never validated in a default build - the descriptor is trusted purely by +//! position, so an image with ordinary rodata there can boot; +//! * what the loader actually reads is min/max_efuse_blk_rev_full at offsets 0xB0/0xB2, and +//! `IS_FIELD_SET()` treats zero as "unset" (bootloader_common_loader.c:102-112). +//! +//! Everything past 0xB5 is reserved padding, so `minimal` stops there and saves 72 bytes of flash. +//! `full` keeps all 256 so that `esptool image-info` can parse it. + +const config = @import("config"); + +pub const magic: u32 = 0xABCD5432; + +pub const Minimal = extern struct { + magic_word: u32 = magic, + secure_version: u32 = 0, + reserv1: [2]u32 = .{ 0, 0 }, + version: [32]u8, + project_name: [32]u8, + time: [16]u8 = @splat(0), + date: [16]u8 = @splat(0), + idf_ver: [32]u8 = @splat(0), + app_elf_sha256: [32]u8 = @splat(0), + min_efuse_blk_rev_full: u16, + max_efuse_blk_rev_full: u16, + mmu_page_size: u8 = 16, // log2(64 KiB); fixed on the P4, which has no configurable MMU page + reserv3: [3]u8 = @splat(0), +}; + +pub const Full = extern struct { + head: Minimal, + reserv2: [18]u32 = @splat(0), +}; + +fn str(comptime n: usize, comptime s: []const u8) [n]u8 { + var out: [n]u8 = @splat(0); + @memcpy(out[0..s.len], s); + return out; +} + +const head: Minimal = .{ + .version = str(32, "0.1"), + .project_name = str(32, "zig-p4"), + // The eFuse *block* revision, which has nothing to do with the silicon revision window that + // -Dmin-rev sets in the image header. Deriving one from the other made `-Dmin-rev=150` emit an + // image the bootloader refuses with "Image requires efuse blk rev >= v0.50". Zero means unset, + // which is what `IS_FIELD_SET()` checks for (bootloader_common_loader.c:102-112). + .min_efuse_blk_rev_full = 0, + .max_efuse_blk_rev_full = 0, +}; + +/// Placed first in .flash.rodata by the linker script, so it lands at image offset 0x20. +/// +/// `export` is load-bearing: a `comptime _ = descriptor;` reference is enough in Debug but not +/// under ReleaseSmall, where the constant is folded away, the section disappears, `KEEP` has +/// nothing to keep, and the image ends up with one mapped segment starting at the wrong offset. +/// An exported symbol is always emitted. +pub export const esp_app_desc linksection(".rodata.appdesc") = if (config.full_descriptor) + Full{ .head = head } +else + head; diff --git a/src/hal.zig b/src/hal.zig new file mode 100644 index 0000000..c62ea2b --- /dev/null +++ b/src/hal.zig @@ -0,0 +1,55 @@ +//! ESP32-P4 hardware abstraction, on top of ESP-IDF's register macros and nothing else. +//! +//! The register *numbers* come from `@import("regs")`, which is `zig translate-c` over ESP-IDF's +//! own `*_reg.h` headers, so no address, shift or mask in this file is a re-derivation of IDF's - +//! they are IDF's, as evaluated by clang. What this file adds is the part macros cannot express: +//! the sequences. Clock-gate before reset, reset before configure, configure before enable; divider +//! arithmetic; which bits must move together in one store. +//! +//! Scope is low-level hardware. Video encode, image signal processing, 2D blitting and the display +//! and camera serial interfaces are deliberately absent; so is anything needing an OS. +//! +//! Every peripheral here is checked against ESP-IDF's own `*_ll.h` implementation of the same +//! operation, in one image, on the die - see `examples/differ.zig` and `tools/differ.zig`. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +pub const Reg = mmio.Reg; +pub const Field = mmio.Field; + +comptime { + // The register set is per-silicon-revision, and the two sets share macro *names* while + // disagreeing on 61 values: AHB_DMA_INFIFO_CNT_CH0 is bits [7:2] under hw_ver1 and [14:8] under + // hw_ver3. A module built from the wrong set is therefore silently wrong rather than absent, so + // the build stamps which one it used and this refuses to compile against anything else. This + // board is a rev v1.3 die, which is what hw_ver1 describes. + if (regs.ZIG_P4_HW_VER != 1) @compileError( + "this HAL targets the pre-v3 ESP32-P4 register set (hw_ver1); rebuild with -Didf-hw-ver=1", + ); +} + +pub const gpio = @import("hal/gpio.zig"); +pub const rwdt = @import("hal/rwdt.zig"); +pub const clkrst = @import("hal/clkrst.zig"); +pub const intr = @import("hal/intr.zig"); +pub const systimer = @import("hal/systimer.zig"); +pub const timg = @import("hal/timg.zig"); +pub const ledc = @import("hal/ledc.zig"); +pub const uart = @import("hal/uart.zig"); +pub const i2c = @import("hal/i2c.zig"); +pub const sdmmc = @import("hal/sdmmc.zig"); + +// Force semantic analysis of every peripheral's decls, not just the ones a test happens to call: +// a wrong `Field.of` pair is a compile error, and an unanalysed function never raises it. +// +// `std.testing.refAllDeclsRecursive` was what this said, and Zig 0.16 does not have it - only +// `refAllDecls` survives (`/usr/lib/zig/std/testing.zig:1213`). Nothing noticed, because no test +// step in build.zig uses this file as a root: `zig build test` roots are tools/image_test.zig, +// src/mmio.zig and the five src/net + src/io files, and `hal` reaches them only as an imported +// *module*, whose tests a test binary does not run. So every `test` in src/hal/*.zig is currently +// compiled by nothing. One line in build.zig's root list fixes that; see this slice's report. +test { + std.testing.refAllDecls(@This()); +} diff --git a/src/hal/clkrst.zig b/src/hal/clkrst.zig new file mode 100644 index 0000000..89a27ef --- /dev/null +++ b/src/hal/clkrst.zig @@ -0,0 +1,265 @@ +//! Peripheral clock gates and resets: HP_SYS_CLKRST. +//! +//! Two things about this block are counter-intuitive on the ESP32-P4, and both were found by +//! reading ESP-IDF rather than by assuming: +//! +//! **Peripheral clocks are already on.** `esp_system/port/soc/esp32p4/clk.c:200` says so in as many +//! words - "All peripheral clocks are default enabled after chip is powered on" - and the reset +//! values in `hp_sys_clkrst_reg.h` agree: REG_UART0_APB_CLK_EN, REG_TIMERGRP0_APB_CLK_EN, +//! REG_SYSTIMER_APB_CLK_EN and REG_IOMUX_APB_CLK_EN all default to 1, with their RST_EN bits at 0. +//! An image that boots from the stock second-stage bootloader never runs `esp_perip_clk_init`, so it +//! inherits those defaults. So this file is not a prerequisite for touching a peripheral; it is what +//! you need to *re*-initialise one, and to reach the few blocks that really are gated off (TWAI is +//! the notable one: REG_TWAI0_APB_CLK_EN defaults to 0). +//! +//! **The hazard is atomicity, not gating.** Every gate and reset bit for the whole chip lives in a +//! handful of shared registers, so `enable(.uart0)` is a read-modify-write of a word that also holds +//! the gate for unrelated peripherals. ESP-IDF makes unguarded calls impossible to compile by +//! referencing `__DECLARE_RCC_ATOMIC_ENV`, an identifier it never defines anywhere; the only legal +//! callers are inside `PERIPH_RCC_ATOMIC()`, which takes a FreeRTOS spinlock. There is no FreeRTOS +//! here and core 1 is held in reset at power-on (`hp_sys_clkrst_reg.h`: REG_RST_EN_CORE1_GLOBAL +//! defaults to 1), so masking interrupts around the read-modify-write is sufficient and is what +//! `atomically` does. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +// The four shared registers this file touches. Which field lives in which register is not derivable +// from the macro names - `HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_S` does not say `SOC_CLK_CTRL2` - so the +// pairing is taken from ESP-IDF's own LL, cited per peripheral below. +const soc_clk_ctrl1 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG); +const soc_clk_ctrl2 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL2_REG); +const soc_clk_ctrl3 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL3_REG); +/// SDMMC's reset bit is not in HP_SYS_CLKRST at all. `sdmmc_ll_reset_register` +/// (`sdmmc_ll.h:158-163`) writes `LP_AON_CLKRST.hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc`, a register +/// of the *low-power* always-on clock-and-reset block, which it shares with the Ethernet MAC. So +/// the `Gates.reset` field is a register as well as a bit, and this is the row that proves it has +/// to be. +const lp_hp_sdmmc_emac_rst_ctrl = Reg.at(regs.LP_CLKRST_HP_SDMMC_EMAC_RST_CTRL_REG); +const hp_rst_en1 = Reg.at(regs.HP_SYS_CLKRST_HP_RST_EN1_REG); + +/// Interrupts masked for the duration of a read-modify-write on a shared register: +/// +/// const guard = clkrst.maskInterrupts(); +/// defer guard.release(); +/// +/// mstatus.MIE is bit 3. `csrrc` clears it and returns the previous mstatus in one instruction, and +/// `release` restores only what was actually there - so this composes: using it inside code that +/// already had interrupts off does not turn them on at the end. +pub const Guard = struct { + prev_mie: bool, + + pub inline fn release(self: Guard) void { + if (self.prev_mie) { + asm volatile ("csrs mstatus, %[mask]" + : + : [mask] "r" (@as(u32, 1 << 3)), + ); + } + } +}; + +pub inline fn maskInterrupts() Guard { + const prev = asm volatile ("csrrc %[out], mstatus, %[mask]" + : [out] "=r" (-> u32), + : [mask] "r" (@as(u32, 1 << 3)), + ); + return .{ .prev_mie = prev & (1 << 3) != 0 }; +} + +/// A peripheral's clock gates and reset bit. +/// +/// `sys_clk` is present only where the peripheral has a second gate on the SYS clock as well as the +/// APB one; UART has both (uart_ll.h:252-253 reads `soc_clk_ctrl2.reg_uart0_apb_clk_en` and +/// `soc_clk_ctrl1.reg_uart0_sys_clk_en`), most blocks have only APB. +const Gates = struct { + apb_clk: ?struct { reg: Reg, field: Field } = null, + sys_clk: ?struct { reg: Reg, field: Field } = null, + reset: struct { reg: Reg, field: Field }, + /// TIMG only: resetting the block re-arms flash-boot protection, which reboots the board a + /// moment later with no diagnostic. `timg_ll.h:53-72` documents it and clears the bit as part of + /// the reset; anything that resets TIMG must do the same. + clears_flashboot: bool = false, +}; + +pub const Peripheral = enum { + uart0, + uart1, + uart2, + uart3, + uart4, + timg0, + timg1, + systimer, + twai0, + ledc, + i2c0, + i2c1, + sdmmc, + + fn gates(comptime self: Peripheral) Gates { + return switch (self) { + // uart_ll.h:251-253 for UART0, and the same three fields per instance after it. + .uart0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART0_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART0_APB_V) }, + }, + .uart1 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART1_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART1_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_V) }, + }, + .uart2 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART2_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART2_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART2_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART2_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART2_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART2_APB_V) }, + }, + .uart3 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART3_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART3_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART3_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART3_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART3_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART3_APB_V) }, + }, + .uart4 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART4_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART4_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART4_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART4_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART4_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART4_APB_V) }, + }, + // timg_ll.h:35-42 for the gate, :60-72 for the reset. The timer groups' APB gate is in + // SOC_CLK_CTRL2 - the same word as the UARTs' - not in PERI_CLK_CTRL21. An earlier + // version of this table had these four entries in PERI_CLK_CTRL21 and so wrote bits + // 21-24 of an unrelated register; hp_sys_clkrst_reg.h:605 defines SOC_CLK_CTRL2_REG and + // :753/:763/:770/:777 put TIMERGRP0 at bit 21, TIMERGRP1 at 22, SYSTIMER at 23 and + // TWAI0 at 24 inside it. PERI_CLK_CTRL20/21 do hold timer-group fields - the per-timer + // clock source and gate, see hal/timg.zig - which is what made the mix-up plausible. + // + // It survived a hardware check because `isClockEnabled` read back the same wrong bit + // `setClockEnabled` had just written: self-consistent, and independent of the chip. + .timg0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP0_S, regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP0_V) }, + .clears_flashboot = true, + }, + .timg1 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP1_S, regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP1_V) }, + .clears_flashboot = true, + }, + // systimer_ll.h:71-72. + .systimer => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_SYSTIMER_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SYSTIMER_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_STIMER_S, regs.HP_SYS_CLKRST_REG_RST_EN_STIMER_V) }, + }, + // The one block whose clock is gated OFF at power-on, which makes it the only peripheral + // where `enable` is observably necessary rather than merely correct. + .twai0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TWAI0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TWAI0_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TWAI0_S, regs.HP_SYS_CLKRST_REG_RST_EN_TWAI0_V) }, + }, + // ledc_ll.h:135 for the gate (`HP_SYS_CLKRST.soc_clk_ctrl3.reg_ledc_apb_clk_en`) and + // :150 for the reset (`hp_rst_en1.reg_rst_en_ledc`). LEDC's APB gate is the *first* bit + // of SOC_CLK_CTRL3, a third register this table did not previously need, and it is one + // of the few whose reset value is 0 (hp_sys_clkrst_reg.h:835): LEDC's registers are + // gated off at power-on, so `setClockEnabled(.ledc, true)` is a prerequisite and not a + // formality. LEDC's *function* clock and its source mux live in PERI_CLK_CTRL22 + // (ledc_ll.h:179, :241) and belong to the peripheral, not to this table - see + // hal/ledc.zig. + .ledc => .{ + .apb_clk = .{ .reg = soc_clk_ctrl3, .field = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_S, regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_V) }, + }, + // i2c_ll.h:149-156 for the gates (`HP_SYS_CLKRST.soc_clk_ctrl2.reg_i2c0_apb_clk_en`, + // and `reg_i2c1_apb_clk_en` for port 1) and :167-176 for the resets + // (`hp_rst_en1.reg_rst_en_i2c0` / `_i2c1`). Both APB gates default to 1 + // (hp_sys_clkrst_reg.h:694-703), so the registers are reachable from boot; what I2C + // does *not* get from this table is its controller clock, whose enable, source mux and + // divider are I2C-specific fields of PERI_CLK_CTRL10/11 and live in hal/i2c.zig. That + // one defaults to 0, so an I2C port brought up through this table alone has readable + // registers and a state machine that never moves. + .i2c0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_S, regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_V) }, + }, + .i2c1 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_I2C1_S, regs.HP_SYS_CLKRST_REG_RST_EN_I2C1_V) }, + }, + // The one row in this table whose two halves live in two different peripherals, and + // the one whose clock really is gated off at power-on alongside LEDC's. + // + // `sdmmc_ll.h:140-144` is the gate: `HP_SYS_CLKRST.soc_clk_ctrl1.reg_sdmmc_sys_clk_en`, + // a *SYS* clock and not an APB one - SDMMC has no APB gate at all, which is why the + // `apb_clk` field is absent here rather than merely unused. It defaults to 0 + // (hp_sys_clkrst_reg.h:475-481, "default: 0"), so `setClockEnabled(.sdmmc, true)` is a + // prerequisite for the register block reading anything but stale values. + // + // `sdmmc_ll.h:158-163` is the reset, and it is in LP_AON_CLKRST: + // `hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc`, bit 28 (lp_clkrst_reg.h:993-999). Looking for + // an `HP_SYS_CLKRST_REG_RST_EN_SDMMC` finds nothing, which is exactly the shape of the + // mistake the timer-group rows above record: a plausible name in the wrong register. + // + // The host clock generator - source mux, divider, sampling phase - is *not* here. It + // is SDMMC-specific and lives in PERI_CLK_CTRL01/02, in hal/sdmmc.zig, the same + // division this table makes for I2C and LEDC. + .sdmmc => .{ + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_V) }, + .reset = .{ .reg = lp_hp_sdmmc_emac_rst_ctrl, .field = Field.of(regs.LP_CLKRST_RST_EN_SDMMC_S, regs.LP_CLKRST_RST_EN_SDMMC_V) }, + }, + }; + } +}; + +/// Turn a peripheral's bus clocks on or off. +pub fn setClockEnabled(comptime p: Peripheral, on: bool) void { + const g = comptime p.gates(); + const v: u32 = @intFromBool(on); + const guard = maskInterrupts(); + defer guard.release(); + if (g.sys_clk) |s| s.reg.modify(.{s.field.is(v)}); + if (g.apb_clk) |a| a.reg.modify(.{a.field.is(v)}); +} + +/// Whether the peripheral's bus clock is on. +/// +/// APB gate if it has one, SYS gate otherwise: SDMMC has only the latter (`sdmmc_ll.h:140-144`), +/// and answering `true` unconditionally for it would have made the oracle's clock check - the one +/// that exists because a gated block reads stale rather than zero - pass on a gated block. +pub fn isClockEnabled(comptime p: Peripheral) bool { + const g = comptime p.gates(); + if (g.apb_clk) |a| return a.reg.get(a.field) == 1; + if (g.sys_clk) |s| return s.reg.get(s.field) == 1; + return true; +} + +/// Pulse a peripheral's reset: assert, deassert. +/// +/// For the timer groups this also clears flash-boot watchdog protection, which the reset re-arms. +/// Leaving that out reboots the board a moment later with nothing on the console to explain it. +pub fn resetPeripheral(comptime p: Peripheral) void { + const g = comptime p.gates(); + { + const guard = maskInterrupts(); + defer guard.release(); + g.reset.reg.modify(.{g.reset.field.is(1)}); + g.reset.reg.modify(.{g.reset.field.is(0)}); + } + if (comptime g.clears_flashboot) { + const wdtconfig0 = Reg.atAddress(switch (p) { + .timg0 => regs.TIMG_WDTCONFIG0_REG(0), + .timg1 => regs.TIMG_WDTCONFIG0_REG(1), + else => unreachable, + }); + wdtconfig0.modify(.{Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V).is(0)}); + } +} + +/// Reset a peripheral and make sure its clocks are on, in that order: a peripheral configured +/// before its reset is released loses the configuration. +pub fn init(comptime p: Peripheral) void { + setClockEnabled(p, true); + resetPeripheral(p); +} diff --git a/src/hal/gpio.zig b/src/hal/gpio.zig new file mode 100644 index 0000000..88a8675 --- /dev/null +++ b/src/hal/gpio.zig @@ -0,0 +1,471 @@ +//! GPIO and the IO MUX. +//! +//! The P4 has 57 pins (GPIO0-56) and every whole-bank register is therefore split in two: `out` +//! covers 0-31 and `out1` covers 32-56. Getting that split wrong is the classic P4 GPIO bug - a +//! write to `out` with a shift of 40 lands on pin 8 - so the bank arithmetic lives in exactly one +//! place here (`Bank`) and every operation goes through it. +//! +//! Levels and enables are driven through the `_W1TS`/`_W1TC` (write-1-to-set / write-1-to-clear) +//! aliases rather than read-modify-write on `out`/`enable`. That is what ESP-IDF's LL does, and it +//! is not a style choice: a read-modify-write of a whole bank races anything else touching another +//! pin in the same bank, and there is no lock here to prevent it. +//! +//! Pad configuration (direction of the *input* buffer, pulls, drive strength, function select) is +//! not in the GPIO peripheral at all - it is in the IO MUX, one register per pad. The two must be +//! kept in step: a pin driven by `enable` but with `fun_ie` clear cannot be read back, which is the +//! single most common "my GPIO does not work" on this part. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// GPIO0-56. 57 pins, and the last five (52-56) exist only on some packages. +pub const max_pin = 56; +pub const pin_count = max_pin + 1; + +/// Which half of a split bank register a pin lives in, and its bit inside that half. +const Bank = struct { + high: bool, + bit: u5, + + inline fn of(pin: u8) Bank { + std.debug.assert(pin <= max_pin); + return if (pin < 32) + .{ .high = false, .bit = @intCast(pin) } + else + .{ .high = true, .bit = @intCast(pin - 32) }; + } + + inline fn mask(self: Bank) u32 { + return @as(u32, 1) << self.bit; + } + + inline fn pick(self: Bank, lo: Reg, hi: Reg) Reg { + return if (self.high) hi else lo; + } +}; + +// The whole-bank registers. `_W1TS`/`_W1TC` are separate addresses that set or clear only the bits +// written as 1, which is what makes a single-pin update atomic against the rest of the bank. +const out = Reg.at(regs.GPIO_OUT_REG); +const out1 = Reg.at(regs.GPIO_OUT1_REG); +const out_w1ts = Reg.at(regs.GPIO_OUT_W1TS_REG); +const out1_w1ts = Reg.at(regs.GPIO_OUT1_W1TS_REG); +const out_w1tc = Reg.at(regs.GPIO_OUT_W1TC_REG); +const out1_w1tc = Reg.at(regs.GPIO_OUT1_W1TC_REG); +const enable_w1ts = Reg.at(regs.GPIO_ENABLE_W1TS_REG); +const enable1_w1ts = Reg.at(regs.GPIO_ENABLE1_W1TS_REG); +const enable_w1tc = Reg.at(regs.GPIO_ENABLE_W1TC_REG); +const enable1_w1tc = Reg.at(regs.GPIO_ENABLE1_W1TC_REG); +const enable = Reg.at(regs.GPIO_ENABLE_REG); +const enable1 = Reg.at(regs.GPIO_ENABLE1_REG); +const in = Reg.at(regs.GPIO_IN_REG); +const in1 = Reg.at(regs.GPIO_IN1_REG); + +/// One IO MUX register per pad, stride taken from two consecutive macros rather than assumed. +const pad = mmio.RegArray( + regs.PERIPHS_IO_MUX_U_PAD_GPIO0, + regs.PERIPHS_IO_MUX_U_PAD_GPIO1, + pin_count, +); + +// Pad fields. These macros are unprefixed globals in io_mux_reg.h - they describe every pad, not +// one - which is why they read as bare `MCU_SEL` rather than `IO_MUX_GPIO7_MCU_SEL`. +const fun_ie = Field.of(regs.FUN_IE_S, regs.FUN_IE_V); +const fun_drv = Field.of(regs.FUN_DRV_S, regs.FUN_DRV_V); +const mcu_sel = Field.of(regs.MCU_SEL_S, regs.MCU_SEL_V); +// io_mux_reg.h defines no macros for the two pull bits; io_mux_struct.h documents them as +// `fun_wpd : R/W; bitpos: [7]` and `fun_wpu : R/W; bitpos: [8]`. +const fun_wpd = Field.bit(7); +const fun_wpu = Field.bit(8); + +/// IO MUX function for a pad. Function 1 is plain GPIO on every P4 pad; the others select a +/// peripheral wired directly to that pad, and anything not on this list has to go through the GPIO +/// matrix instead. +pub const Function = enum(u3) { + f0 = 0, + /// Plain GPIO - the GPIO peripheral drives and samples the pad. + gpio = 1, + f2 = 2, + f3 = 3, + f4 = 4, + f5 = 5, + f6 = 6, + f7 = 7, +}; + +pub const Drive = enum(u2) { + /// ~5 mA + weakest = 0, + /// ~10 mA + weak = 1, + /// ~20 mA, the reset value + medium = 2, + /// ~40 mA + strong = 3, +}; + +pub const Pull = enum { none, up, down }; + +// ------------------------------------------------------------------------------------- levels + +/// Drive a pin high or low. Uses the write-1-to-set/clear alias, so no other pin in the bank is +/// disturbed and no read is needed. +pub inline fn setLevel(pin: u8, level: u1) void { + const b = Bank.of(pin); + const r = if (level == 1) + b.pick(out_w1ts, out1_w1ts) + else + b.pick(out_w1tc, out1_w1tc); + r.writeRaw(b.mask()); +} + +pub inline fn setHigh(pin: u8) void { + setLevel(pin, 1); +} + +pub inline fn setLow(pin: u8) void { + setLevel(pin, 0); +} + +pub inline fn toggle(pin: u8) void { + const b = Bank.of(pin); + if (b.pick(out, out1).raw() & b.mask() != 0) setLow(pin) else setHigh(pin); +} + +/// Sample the pad. Reads the *input* register, so it reports what the pin is actually at - which +/// for an open-drain or externally driven pin is not necessarily what was last written to `out`. +/// Requires the pad's input buffer to be enabled (`setInputEnable`). +pub inline fn getLevel(pin: u8) u1 { + const b = Bank.of(pin); + return @intCast((b.pick(in, in1).raw() >> b.bit) & 1); +} + +/// What was last driven, from the output register rather than the pad. +pub inline fn getDrivenLevel(pin: u8) u1 { + const b = Bank.of(pin); + return @intCast((b.pick(out, out1).raw() >> b.bit) & 1); +} + +// -------------------------------------------------------------------------------- direction + +pub inline fn outputEnable(pin: u8) void { + const b = Bank.of(pin); + b.pick(enable_w1ts, enable1_w1ts).writeRaw(b.mask()); +} + +pub inline fn outputDisable(pin: u8) void { + const b = Bank.of(pin); + b.pick(enable_w1tc, enable1_w1tc).writeRaw(b.mask()); +} + +pub inline fn isOutputEnabled(pin: u8) bool { + const b = Bank.of(pin); + return b.pick(enable, enable1).raw() & b.mask() != 0; +} + +/// The pad's input buffer. Independent of the output driver: both can be on at once, which is how a +/// pin is read back while being driven. +pub inline fn setInputEnable(pin: u8, on: bool) void { + pad.at(pin).modify(.{fun_ie.is(@intFromBool(on))}); +} + +/// Whether the pad's input buffer is on. The counterpart of `setInputEnable`, and worth having +/// because a routed input with `fun_ie` clear is indistinguishable from a card that never drove +/// the pin: both read as a constant. +pub inline fn isInputEnabled(pin: u8) bool { + return pad.at(pin).get(fun_ie) != 0; +} + +// -------------------------------------------------------------------------------- pad config + +pub inline fn setFunction(pin: u8, f: Function) void { + pad.at(pin).modify(.{mcu_sel.is(@intFromEnum(f))}); +} + +pub inline fn setDrive(pin: u8, d: Drive) void { + pad.at(pin).modify(.{fun_drv.is(@intFromEnum(d))}); +} + +/// Internal pull resistors. Setting one direction always clears the other in the same store: a pad +/// with both enabled is a fight between two resistors, and it is easy to reach by two calls. +pub inline fn setPull(pin: u8, p: Pull) void { + pad.at(pin).modify(.{ + fun_wpu.is(@intFromBool(p == .up)), + fun_wpd.is(@intFromBool(p == .down)), + }); +} + +/// What `setPull` last left, read back from the pad. A pad with both resistors enabled cannot be +/// reached through `setPull`, but the reset value or another driver can leave one that way, so the +/// contradictory case is reported as `.none` rather than picking a winner. +pub inline fn getPull(pin: u8) Pull { + const w = pad.at(pin).raw(); + const up = w & fun_wpu.mask() != 0; + const down = w & fun_wpd.mask() != 0; + if (up and !down) return .up; + if (down and !up) return .down; + return .none; +} + +/// Open-drain: the pad drives low and releases high instead of driving both rails. +/// +/// This one is not in the IO MUX with the other pad properties - it is `GPIO_PINn_PAD_DRIVER`, bit +/// 2 of the GPIO peripheral's per-pin register (`gpio_reg.h:363-368`, "1:open-drain. 0:normal"), +/// which is a different register file from `PERIPHS_IO_MUX_U_PAD_GPIOn`. A shared bus - I2C, or any +/// wired-AND signal - needs this on both pads *and* an external pull-up; the internal pull-up is +/// too weak for anything but a short trace at a low bit rate. +pub inline fn setOpenDrain(pin: u8, on: bool) void { + pin_cfg.at(pin).modify(.{pad_driver.is(@intFromBool(on))}); +} + +/// The GPIO peripheral's per-pin configuration register, one per pad. Not the IO MUX: this file +/// holds the open-drain select, the interrupt configuration and the input synchroniser bypasses. +const pin_cfg = mmio.RegArray(regs.GPIO_PIN0_REG, regs.GPIO_PIN1_REG, pin_count); +const pad_driver = Field.of(regs.GPIO_PIN0_PAD_DRIVER_S, regs.GPIO_PIN0_PAD_DRIVER_V); + +// -------------------------------------------------------------------------- pin interrupts + +/// How a pad raises its interrupt. `gpio_reg.h:377-381`: "0:disable GPIO interrupt. 1:trigger at +/// posedge. 2:trigger at negedge. 3:trigger at any edge. 4:valid at low level. 5:valid at high +/// level". +pub const IntrType = enum(u3) { + disable = 0, + posedge = 1, + negedge = 2, + anyedge = 3, + low_level = 4, + high_level = 5, +}; + +const int_type = Field.of(regs.GPIO_PIN0_INT_TYPE_S, regs.GPIO_PIN0_INT_TYPE_V); +/// Five bits, one per consumer of the pad's interrupt, not a boolean. `gpio_reg.h:400-402` says +/// "set bit 13 to enable CPU interrupt, set bit 14 to enable CPU(not shielded) interrupt", and +/// `gpio_ll.h:41,213` names bit 0 of the field `GPIO_LL_INTR0_ENA` and writes exactly that to +/// route a pad to the `gpio_intr0` source. Writing 1 here means "line 0", not "enabled". +const int_ena = Field.of(regs.GPIO_PIN0_INT_ENA_S, regs.GPIO_PIN0_INT_ENA_V); + +/// Which of the P4's four GPIO interrupt outputs a pad drives. Each is a separate entry in the +/// interrupt matrix (`hal.intr.Source.gpio_intr0` .. `gpio_intr3`), and each has its own status +/// register pair. ESP-IDF only ever uses line 0 - `gpio_ll_intr_enable_on_core` hard-codes +/// `GPIO_LL_INTR0_ENA` with a "TODO: IDF-7995" beside it - so line 0 is the tested path. +pub const IntrLine = enum(u3) { + line0 = 0, + line1 = 1, + line2 = 2, + line3 = 3, +}; + +/// Per-line status, gated by `int_ena`. Reading `status`/`status1` instead would report pads whose +/// interrupt is configured but routed to a different line. `gpio_reg.h:277,291` for line 0, +/// `:302,316` for line 1; lines 2 and 3 continue the same +0x8 stride. +const intr_status = mmio.RegArray(regs.GPIO_INTR_0_REG, regs.GPIO_INTR_1_REG, 4); +const intr_status1 = mmio.RegArray(regs.GPIO_INTR1_0_REG, regs.GPIO_INTR1_1_REG, 4); + +/// Status is cleared through a shared write-1-to-clear register, not a per-line one: one pad has +/// one latch however many lines observe it. `gpio_reg.h:233,269`. +const status_w1tc = Reg.at(regs.GPIO_STATUS_W1TC_REG); +const status1_w1tc = Reg.at(regs.GPIO_STATUS1_W1TC_REG); + +/// Arm a pad's interrupt and route it to one of the four GPIO interrupt outputs. +/// +/// This is the GPIO peripheral's half only. The other half is `hal.intr`: the chosen line still +/// has to be routed from `Source.gpio_intr0`+n to a CLIC line and given a handler. Doing it in two +/// calls is deliberate - one pad's interrupt and one CPU line are not the same resource, and +/// several pads normally share a line. +/// +/// Stale latched status is cleared first. A pad that saw an edge before its interrupt was armed +/// otherwise fires immediately on enable, which looks exactly like a real event. +pub fn setInterrupt(pin: u8, t: IntrType, line: IntrLine) void { + std.debug.assert(pin <= max_pin); + clearInterrupt(pin); + pin_cfg.at(pin).modify(.{ + int_type.is(@intFromEnum(t)), + int_ena.is(if (t == .disable) 0 else @as(u32, 1) << @intFromEnum(line)), + }); +} + +/// Disarm, leaving the trigger type alone so it can be re-enabled unchanged. +pub fn disableInterrupt(pin: u8) void { + pin_cfg.at(pin).modify(.{int_ena.is(0)}); +} + +pub fn interruptPending(pin: u8, line: IntrLine) bool { + const b = Bank.of(pin); + const i: u32 = @intFromEnum(line); + return b.pick(intr_status.at(i), intr_status1.at(i)).raw() & b.mask() != 0; +} + +/// Every pad currently interrupting on `line`, as a 57-bit mask in two halves. One read of each +/// register, so a handler can dispatch the whole set without re-reading between pads. +pub fn pendingMask(line: IntrLine) struct { low: u32, high: u32 } { + const i: u32 = @intFromEnum(line); + return .{ .low = intr_status.at(i).raw(), .high = intr_status1.at(i).raw() }; +} + +pub fn clearInterrupt(pin: u8) void { + const b = Bank.of(pin); + b.pick(status_w1tc, status1_w1tc).writeRaw(b.mask()); +} + +pub fn clearInterrupts(low: u32, high: u32) void { + if (low != 0) status_w1tc.writeRaw(low); + if (high != 0) status1_w1tc.writeRaw(high); +} + +/// Everything a pin needs to be a plain push-pull output, in the order the hardware wants: select +/// the pad's function before enabling the driver, so the pin never spends a moment driven by +/// whatever peripheral the IO MUX happened to be pointing at. +pub fn configureOutput(pin: u8, opts: struct { + drive: Drive = .medium, + /// Enable the input buffer too, so the pin can be read back. + readback: bool = false, +}) void { + setFunction(pin, .gpio); + // Point the matrix at the GPIO peripheral: a pad left routed to whatever signal was there + // before is the failure this line prevents. + func_out_sel.at(pin).modify(.{ out_sel.is(matrix_gpio_signal), oen_sel.is(0) }); + pad.at(pin).modify(.{ + fun_drv.is(@intFromEnum(opts.drive)), + fun_ie.is(@intFromBool(opts.readback)), + fun_wpu.is(0), + fun_wpd.is(0), + }); + outputEnable(pin); +} + +/// A plain input: driver off, input buffer on, optional pull. +pub fn configureInput(pin: u8, opts: struct { pull: Pull = .none }) void { + outputDisable(pin); + setFunction(pin, .gpio); + pad.at(pin).modify(.{ + fun_ie.is(1), + fun_wpu.is(@intFromBool(opts.pull == .up)), + fun_wpd.is(@intFromBool(opts.pull == .down)), + }); +} + +// ------------------------------------------------------------------------------- GPIO matrix + +/// The GPIO matrix: 256 peripheral output signals, any of which can be routed to any pad. This is +/// how a UART reaches a pin that has no direct IO MUX function for it. +const func_out_sel = mmio.RegArray( + regs.GPIO_FUNC0_OUT_SEL_CFG_REG, + regs.GPIO_FUNC1_OUT_SEL_CFG_REG, + pin_count, +); +// The input side of the matrix, indexed by *signal* rather than by pad: GPIO_FUNCn_IN_SEL_CFG +// selects which pad feeds peripheral input signal n. That is the opposite indexing from +// `func_out_sel` above, and it is why the two arrays exist separately. +// +// The base is FUNC1's address minus one word, not FUNC1's address. gpio_struct.h:849 declares +// `func_in_sel_cfg[256]` and notes func0 is reserved, so ESP-IDF's register header defines no +// GPIO_FUNC0_IN_SEL_CFG_REG at all - the array starts at +0x158 with a name-less word. Anchoring +// on FUNC1 with a count of 256 is off by one in both directions: `at(n)` would configure signal +// n+1, and `at(255)` would land on GPIO_FUNC0_OUT_SEL_CFG_REG (+0x558) and start driving a pad. +// Bounds checked against the headers: FUNC255_IN_SEL_CFG_REG is +0x554 = 0x158 + 4*255. +const func_in_sel = mmio.RegArray( + regs.GPIO_FUNC1_IN_SEL_CFG_REG - 4, + regs.GPIO_FUNC1_IN_SEL_CFG_REG, + 256, +); + +const out_sel = Field.of(regs.GPIO_FUNC0_OUT_SEL_S, regs.GPIO_FUNC0_OUT_SEL_V); +const oen_sel = Field.of(regs.GPIO_FUNC0_OEN_SEL_S, regs.GPIO_FUNC0_OEN_SEL_V); + +// The input side's three fields. All of GPIO_FUNCn_IN_SEL_CFG's fields share these shifts, so as +// with the pad registers one macro triple describes all 256. +const in_sel = Field.of(regs.GPIO_FUNC1_IN_SEL_S, regs.GPIO_FUNC1_IN_SEL_V); +const in_inv_sel = Field.of(regs.GPIO_FUNC1_IN_INV_SEL_S, regs.GPIO_FUNC1_IN_INV_SEL_V); +/// 1 = take this signal from the GPIO matrix, 0 = from the pad's direct IO MUX function. +const sig_in_sel = Field.of(regs.GPIO_SIG1_IN_SEL_S, regs.GPIO_SIG1_IN_SEL_V); + +/// Writing this value instead of a peripheral signal index means "the GPIO peripheral drives this +/// pad", which is the matrix's way of expressing plain GPIO output. It comes from IDF's own signal +/// map because it is chip-specific: 256 here, 128 on the ESP32-S3. +pub const matrix_gpio_signal: u32 = regs.SIG_GPIO_OUT_IDX; + +/// Route a peripheral output signal to a pad through the matrix, and let that peripheral own the +/// pad's output enable. +/// +/// `OEN_SEL` reads backwards from its name, and the differential test against ESP-IDF's LL is what +/// caught it: 1 means "use GPIO_ENABLE_REG[n] as the output enable", 0 means "use the peripheral's +/// own output enable signal" (gpio_reg.h, GPIO_FUNC0_OEN_SEL). A routed peripheral must have 0 - its +/// OE is part of the signal being routed. The first version of this function set 1 and then set the +/// matching GPIO_ENABLE bit to compensate, which worked by the wrong mechanism and left the pad +/// latently output-enabled: clear OEN_SEL later and the pin would start driving on its own. +pub fn matrixOut(pin: u8, signal: u32) void { + std.debug.assert(pin <= max_pin); + setFunction(pin, .gpio); + func_out_sel.at(pin).modify(.{ out_sel.is(signal), oen_sel.is(0) }); +} + +/// Route a pad to a peripheral *input* signal through the matrix. +/// +/// Indexed by signal, not by pin, which is the opposite of `matrixOut`: one pad may feed any number +/// of input signals, but a signal has exactly one source. The three writes are one word, where +/// gpio_ll.h:613-618 uses three bitfield stores; the resulting word is identical and nothing here +/// depends on the intermediate states, whereas a driver that read the register back between them +/// could observe a signal sourced from the wrong pad. +/// +/// This does not enable the pad's input buffer - `setInputEnable` does, and a routed input with +/// `fun_ie` clear reads as a constant. Callers that want the pad readable must do both. +pub fn matrixIn(pin: u8, signal: u32) void { + std.debug.assert(pin <= max_pin or pin == matrix_const_zero or pin == matrix_const_one); + std.debug.assert(signal < 256); + func_in_sel.at(signal).modify(.{ + in_sel.is(pin), + in_inv_sel.is(0), + sig_in_sel.is(1), + }); +} + +/// Where a peripheral input signal is sourced from. The read side of `matrixIn`, for a diagnostic +/// that has to distinguish "routed to the wrong pad" from "not routed at all" - the two look the +/// same from the peripheral's end. +pub const MatrixIn = struct { + /// A pad index, or `matrix_const_zero`/`matrix_const_one`. Meaningless when `from_matrix` is + /// false: the field keeps its reset value in that case, which can read like a deliberate + /// tie-high and is not one. + pin: u8, + inverted: bool, + /// `sig_in_sel`. False means the matrix is bypassed entirely and the signal comes from the + /// pad's direct IO MUX function - which for a peripheral that has none is undefined. + from_matrix: bool, +}; + +pub fn matrixInSource(signal: u32) MatrixIn { + std.debug.assert(signal < 256); + const w = func_in_sel.at(signal).raw(); + return .{ + .pin = @intCast((w >> in_sel.shift) & in_sel.unshiftedMask()), + .inverted = w & in_inv_sel.mask() != 0, + .from_matrix = w & sig_in_sel.mask() != 0, + }; +} + +/// Two values of `matrixIn`'s `pin` that are not pins: they tie the signal to a constant level +/// inside the matrix. `gpio_reg.h:3717-3719` documents the encoding on the register itself - +/// "s=0-56: connect GPIO[s] to this port. s=0x3F: set this port always high level. s=0x3E: set +/// this port always low level" - and `soc/gpio_pins.h:13-14` gives them the names ESP-IDF's +/// drivers use. They are chip-specific: 0x38/0x30 on the ESP32, 0x1E/0x1F on the C3. +/// +/// This is how an unwired peripheral input gets a defined level. Leaving one alone is not +/// equivalent: `in_sel` does default to 0x3F, but `sig_in_sel` defaults to 0, which bypasses the +/// matrix entirely and takes the signal from the pad's direct IO MUX function - which for a +/// peripheral that has none is not a constant anything. `hal/sdmmc.zig` needs both of these for +/// slot 1's card-detect and card-interrupt inputs. +pub const matrix_const_one: u8 = 0x3f; +pub const matrix_const_zero: u8 = 0x3e; + +test "bank arithmetic splits at 32, which is where the P4's second register begins" { + try std.testing.expectEqual(@as(u5, 20), Bank.of(20).bit); + try std.testing.expect(!Bank.of(20).high); + try std.testing.expectEqual(@as(u5, 0), Bank.of(32).bit); + try std.testing.expect(Bank.of(32).high); + try std.testing.expectEqual(@as(u5, 24), Bank.of(56).bit); + try std.testing.expectEqual(@as(u32, 1) << 24, Bank.of(56).mask()); +} diff --git a/src/hal/i2c.zig b/src/hal/i2c.zig new file mode 100644 index 0000000..2a1d8db --- /dev/null +++ b/src/hal/i2c.zig @@ -0,0 +1,1091 @@ +//! I2C0 and I2C1 in master mode, FIFO access, no interrupts and no DMA. +//! +//! Slave mode, LP_I2C and the RAM (non-FIFO) access path are deliberately absent. +//! +//! Three things about this peripheral are not visible in the register headers, and each one is a +//! way for a port to produce a bus that half-works: +//! +//! **1. The timing is a dozen registers computed from one number.** SCL low, SCL high, SCL +//! wait-high, SDA hold, SDA sample, start hold, restart setup, stop hold, stop setup and the +//! timeout exponent all come from a single `half_cycle` derived from the source clock and the wanted +//! SCL frequency, and several of them are written *minus one* while two deliberately are not. The +//! arithmetic is reproduced from ESP-IDF exactly, with the line numbers, in `Timing.calculate` and +//! `applyTiming` below - including the parts that look like bugs and are not. +//! +//! **2. Nothing takes effect until `CONF_UPGATE` is written.** The timing and control registers feed +//! a synchroniser rather than the state machine directly, so a driver that configures the block and +//! starts a transaction without `commitConfig()` runs on the *previous* configuration. It is a +//! write-to-trigger bit that reads back 0, so nothing about the register state afterwards shows +//! whether it was ever written - which is exactly the kind of bug a state-comparing differential +//! test cannot see, so it is called out here instead. ESP-IDF puts the call in the driver +//! (`esp_driver_i2c/i2c_master.c:96`, `i2c_ll_update` at `i2c_ll.h:137-141`), not in the LL +//! functions that write the timing. +//! +//! **3. The command opcode numbers changed after the original ESP32, and this chip's own register +//! header still documents the old ones.** `i2c_struct.h:1009-1021` and `i2c_reg.h:1174-1186` say +//! "0: RSTART, 1: WRITE, 2: READ, 3: STOP, 4: END". ESP-IDF's P4 LL says RESTART=6, WRITE=1, +//! READ=3, STOP=2 (`i2c_ll.h:55-59`), which is what every post-ESP32 target uses (esp32c3, esp32c6 +//! and esp32p4 agree; only `esp32/include/hal/i2c_ll.h:47-51` has the numbers the P4 header's prose +//! describes). The LL is the version the shipping driver runs on silicon, so it is the one here, and +//! `i2c_ref.c` builds its command words from IDF's own `I2C_LL_CMD_*` macros so that the +//! differential test would catch a wrong constant here rather than agreeing with it. +//! +//! And one hazard for anything that snapshots this block: **reading `I2C_DATA_REG` pops the RX +//! FIFO.** See `Data register` below. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// HP I2C instances. LP_I2C is a third `i2c_dev_t` in ESP-IDF (`SOC_I2C_NUM` is 3, `soc_caps.h:310`) +/// but it lives in the LP domain with its own clock and pad rules, and is out of scope here. +pub const port_count: u8 = 2; + +/// Bytes in each direction. `i2c_ll.h:29` (`I2C_LL_FIFO_LEN`); the RAM behind it is 32 bytes at +/// +0x100 (TX) and +0x180 (RX), reachable directly only in non-FIFO mode. +pub const fifo_len: u8 = 32; + +/// Command slots. **Eight on this chip**, not sixteen: `i2c_ll.h:31` says `I2C_LL_CMD_REG_NUM 8`, +/// `i2c_struct.h:1073` declares `command[8]`, and the register header stops at `I2C_COMD7_REG` +/// (+0x74). ESP-IDF's own `i2c_ll_master_write_cmd_reg` doc comment claims "should be less than 16" +/// (`i2c_ll.h:433`) - that comment is stale, and `i2c_ll_master_is_cmd_done` two hundred lines later +/// says 8 (`i2c_ll.h:1043`). Eight slots is why the driver's long transfers end a chunk with an END +/// opcode and continue: there is no room for a command per byte. +pub const cmd_slots: u8 = 8; + +// ------------------------------------------------------------------------------------ registers +// +// One array per register, indexed by port. The stride is checked against I2C1's own macro rather +// than assumed: `REG_I2C_BASE(i)` is `DR_REG_I2C0_BASE + i * 0x1000` (`soc/esp32p4/include/soc/ +// soc.h:24`), which the linker script agrees with (`esp32p4.peripherals.ld:17-18`, I2C0 = +// 0x500C4000, I2C1 = 0x500C5000). + +fn portArray(comptime macro0: anytype, comptime macro1: anytype) type { + return mmio.RegArray(macro0, macro1, port_count); +} + +const scl_low_period = portArray(regs.I2C_SCL_LOW_PERIOD_REG(0), regs.I2C_SCL_LOW_PERIOD_REG(1)); +const ctr = portArray(regs.I2C_CTR_REG(0), regs.I2C_CTR_REG(1)); +const sr = portArray(regs.I2C_SR_REG(0), regs.I2C_SR_REG(1)); +const to = portArray(regs.I2C_TO_REG(0), regs.I2C_TO_REG(1)); +const fifo_st = portArray(regs.I2C_FIFO_ST_REG(0), regs.I2C_FIFO_ST_REG(1)); +const fifo_conf = portArray(regs.I2C_FIFO_CONF_REG(0), regs.I2C_FIFO_CONF_REG(1)); +const data = portArray(regs.I2C_DATA_REG(0), regs.I2C_DATA_REG(1)); +const int_raw = portArray(regs.I2C_INT_RAW_REG(0), regs.I2C_INT_RAW_REG(1)); +const int_clr = portArray(regs.I2C_INT_CLR_REG(0), regs.I2C_INT_CLR_REG(1)); +const int_ena = portArray(regs.I2C_INT_ENA_REG(0), regs.I2C_INT_ENA_REG(1)); +const sda_hold = portArray(regs.I2C_SDA_HOLD_REG(0), regs.I2C_SDA_HOLD_REG(1)); +const sda_sample = portArray(regs.I2C_SDA_SAMPLE_REG(0), regs.I2C_SDA_SAMPLE_REG(1)); +const scl_high_period = portArray(regs.I2C_SCL_HIGH_PERIOD_REG(0), regs.I2C_SCL_HIGH_PERIOD_REG(1)); +const scl_start_hold = portArray(regs.I2C_SCL_START_HOLD_REG(0), regs.I2C_SCL_START_HOLD_REG(1)); +const scl_rstart_setup = portArray(regs.I2C_SCL_RSTART_SETUP_REG(0), regs.I2C_SCL_RSTART_SETUP_REG(1)); +const scl_stop_hold = portArray(regs.I2C_SCL_STOP_HOLD_REG(0), regs.I2C_SCL_STOP_HOLD_REG(1)); +const scl_stop_setup = portArray(regs.I2C_SCL_STOP_SETUP_REG(0), regs.I2C_SCL_STOP_SETUP_REG(1)); +const filter_cfg = portArray(regs.I2C_FILTER_CFG_REG(0), regs.I2C_FILTER_CFG_REG(1)); +const comd0 = portArray(regs.I2C_COMD0_REG(0), regs.I2C_COMD0_REG(1)); +const scl_sp_conf = portArray(regs.I2C_SCL_SP_CONF_REG(0), regs.I2C_SCL_SP_CONF_REG(1)); + +/// First address of a port's register block, for the differential harness's window. +pub inline fn base(port: u8) u32 { + std.debug.assert(port < port_count); + return @intCast(scl_low_period.base + scl_low_period.stride * port); +} + +// I2C_CTR_REG. `trans_start`, `fsm_rst` and `conf_upgate` are write-to-trigger: they read back 0, +// so a read-modify-write of this register does not re-trigger them. +const sda_force_out = Field.of(regs.I2C_SDA_FORCE_OUT_S, regs.I2C_SDA_FORCE_OUT_V); +const scl_force_out = Field.of(regs.I2C_SCL_FORCE_OUT_S, regs.I2C_SCL_FORCE_OUT_V); +const rx_full_ack_level = Field.of(regs.I2C_RX_FULL_ACK_LEVEL_S, regs.I2C_RX_FULL_ACK_LEVEL_V); +const ms_mode = Field.of(regs.I2C_MS_MODE_S, regs.I2C_MS_MODE_V); +const trans_start = Field.of(regs.I2C_TRANS_START_S, regs.I2C_TRANS_START_V); +const tx_lsb_first = Field.of(regs.I2C_TX_LSB_FIRST_S, regs.I2C_TX_LSB_FIRST_V); +const rx_lsb_first = Field.of(regs.I2C_RX_LSB_FIRST_S, regs.I2C_RX_LSB_FIRST_V); +const arbitration_en = Field.of(regs.I2C_ARBITRATION_EN_S, regs.I2C_ARBITRATION_EN_V); +const fsm_rst = Field.of(regs.I2C_FSM_RST_S, regs.I2C_FSM_RST_V); +const conf_upgate = Field.of(regs.I2C_CONF_UPGATE_S, regs.I2C_CONF_UPGATE_V); + +// I2C_SR_REG, all read-only. +const resp_rec = Field.of(regs.I2C_RESP_REC_S, regs.I2C_RESP_REC_V); +const arb_lost = Field.of(regs.I2C_ARB_LOST_S, regs.I2C_ARB_LOST_V); +const bus_busy = Field.of(regs.I2C_BUS_BUSY_S, regs.I2C_BUS_BUSY_V); +const rxfifo_cnt = Field.of(regs.I2C_RXFIFO_CNT_S, regs.I2C_RXFIFO_CNT_V); +const txfifo_cnt = Field.of(regs.I2C_TXFIFO_CNT_S, regs.I2C_TXFIFO_CNT_V); + +// I2C_TO_REG. `time_out_value` is only five bits wide - the timeout is 2^value source-clock cycles, +// so 31 is the largest legal exponent and the arithmetic below never approaches it. +const time_out_value = Field.of(regs.I2C_TIME_OUT_VALUE_S, regs.I2C_TIME_OUT_VALUE_V); +const time_out_en = Field.of(regs.I2C_TIME_OUT_EN_S, regs.I2C_TIME_OUT_EN_V); + +// I2C_FIFO_CONF_REG. `rx_fifo_rst`/`tx_fifo_rst` are annotated R/W, not self-clearing: they hold +// the FIFO in reset until written back to 0, which is why resetting one is two stores. +const rxfifo_wm_thrhd = Field.of(regs.I2C_RXFIFO_WM_THRHD_S, regs.I2C_RXFIFO_WM_THRHD_V); +const txfifo_wm_thrhd = Field.of(regs.I2C_TXFIFO_WM_THRHD_S, regs.I2C_TXFIFO_WM_THRHD_V); +const nonfifo_en = Field.of(regs.I2C_NONFIFO_EN_S, regs.I2C_NONFIFO_EN_V); +const rx_fifo_rst = Field.of(regs.I2C_RX_FIFO_RST_S, regs.I2C_RX_FIFO_RST_V); +const tx_fifo_rst = Field.of(regs.I2C_TX_FIFO_RST_S, regs.I2C_TX_FIFO_RST_V); +const fifo_prt_en = Field.of(regs.I2C_FIFO_PRT_EN_S, regs.I2C_FIFO_PRT_EN_V); + +// Timing fields. Every period is nine bits ([8:0], max 511) except `scl_wait_high_period`, which is +// seven ([15:9], max 127) and shares its register with `scl_high_period`. +const scl_low_period_f = Field.of(regs.I2C_SCL_LOW_PERIOD_S, regs.I2C_SCL_LOW_PERIOD_V); +const scl_high_period_f = Field.of(regs.I2C_SCL_HIGH_PERIOD_S, regs.I2C_SCL_HIGH_PERIOD_V); +const scl_wait_high_period_f = Field.of(regs.I2C_SCL_WAIT_HIGH_PERIOD_S, regs.I2C_SCL_WAIT_HIGH_PERIOD_V); +const sda_hold_time = Field.of(regs.I2C_SDA_HOLD_TIME_S, regs.I2C_SDA_HOLD_TIME_V); +const sda_sample_time = Field.of(regs.I2C_SDA_SAMPLE_TIME_S, regs.I2C_SDA_SAMPLE_TIME_V); +const scl_start_hold_time = Field.of(regs.I2C_SCL_START_HOLD_TIME_S, regs.I2C_SCL_START_HOLD_TIME_V); +const scl_rstart_setup_time = Field.of(regs.I2C_SCL_RSTART_SETUP_TIME_S, regs.I2C_SCL_RSTART_SETUP_TIME_V); +const scl_stop_hold_time = Field.of(regs.I2C_SCL_STOP_HOLD_TIME_S, regs.I2C_SCL_STOP_HOLD_TIME_V); +const scl_stop_setup_time = Field.of(regs.I2C_SCL_STOP_SETUP_TIME_S, regs.I2C_SCL_STOP_SETUP_TIME_V); + +// I2C_FILTER_CFG_REG. Both thresholds are four bits, both filters default *enabled* with a +// threshold of 0 - which filters nothing - so "disable" and "enable with 0" are different words. +const scl_filter_thres = Field.of(regs.I2C_SCL_FILTER_THRES_S, regs.I2C_SCL_FILTER_THRES_V); +const sda_filter_thres = Field.of(regs.I2C_SDA_FILTER_THRES_S, regs.I2C_SDA_FILTER_THRES_V); +const scl_filter_en = Field.of(regs.I2C_SCL_FILTER_EN_S, regs.I2C_SCL_FILTER_EN_V); +const sda_filter_en = Field.of(regs.I2C_SDA_FILTER_EN_S, regs.I2C_SDA_FILTER_EN_V); + +// I2C_SCL_SP_CONF_REG: the hardware bus-clear generator. +const scl_rst_slv_en = Field.of(regs.I2C_SCL_RST_SLV_EN_S, regs.I2C_SCL_RST_SLV_EN_V); +const scl_rst_slv_num = Field.of(regs.I2C_SCL_RST_SLV_NUM_S, regs.I2C_SCL_RST_SLV_NUM_V); + +/// Data register offset in words, for the harness's `no_read` list. See `Data register` below. +pub const data_word_offset: u32 = (0x1c - 0x00) / 4; + +// ----------------------------------------------------------------------------- clocks and reset +// +// I2C has clock control in two places, and the split is not symmetrical between the two ports: +// +// * the APB bus clock gate and the block reset are in HP_SYS_CLKRST's shared registers, and live +// in `clkrst.zig` with every other peripheral's (`i2c_ll.h:149-176`); +// * the *controller* clock - the one the bus state machine runs on - its source select and its +// divider are I2C-specific fields of HP_SYS_CLKRST_PERI_CLK_CTRL10/11, and are here. +// +// The asymmetry is the trap: I2C1's source select and controller-clock enable are in PERI_CLK_CTRL10 +// beside I2C0's (bits 26 and 27, `i2c_ll.h:851-852` and `i2c_ll.h:944-945`), while I2C1's *divider* +// is in PERI_CLK_CTRL11 (`i2c_ll.h:196-199`). Reading the field names alone would put all of I2C1 +// in ctrl11. + +const peri_clk_ctrl10 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL10_REG); +const peri_clk_ctrl11 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL11_REG); + +const i2c0_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_V); +const i2c1_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_V); +const i2c0_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_V); +const i2c1_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_V); +const i2c0_div_num = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_V); +const i2c0_div_numerator = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_V); +const i2c0_div_denominator = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_V); +const i2c1_div_num = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_V); +const i2c1_div_numerator = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_V); +const i2c1_div_denominator = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_V); + +/// Controller clock source. Two choices on this chip (`clk_tree_defs.h:486-494`), and the register +/// field is one bit: 0 = XTAL, 1 = RC_FAST (`i2c_ll.h:848-852`). +pub const Source = enum(u1) { + /// 40 MHz on this board, and the default. Accurate, which for a bus with a specified maximum + /// clock is the whole point. + xtal = 0, + /// The internal RC oscillator, ~20 MHz and temperature-dependent. Usable only because I2C is a + /// clocked bus with no baud-rate agreement to keep. + rc_fast = 1, +}; + +/// XTAL frequency on this board, as the source frequency to hand `Timing.calculate` for +/// `Source.xtal`. Fixed by the crystal, not by the clock tree: 40 MHz. +pub const xtal_hz: u32 = 40_000_000; + +/// Select the controller clock source. A read-modify-write of a register shared with the other +/// port's clock fields, so it takes the interrupt guard. +pub fn setSource(port: u8, src: Source) void { + std.debug.assert(port < port_count); + const v: u32 = @intFromEnum(src); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl10.modify(.{if (port == 0) i2c0_clk_src_sel.is(v) else i2c1_clk_src_sel.is(v)}); +} + +/// The controller clock gate, which is *not* the APB gate in `clkrst.zig`: registers stay readable +/// and writable with this off, and only the bus state machine stops. It defaults to 0 +/// (`hp_sys_clkrst_reg.h`, REG_I2C0_CLK_EN default 0), so unlike most peripherals on this chip I2C +/// genuinely needs this call before it will do anything. `_i2c_hal_init` (`i2c_hal.c:52-58`) is +/// where ESP-IDF makes it. +pub fn setControllerClockEnabled(port: u8, on: bool) void { + std.debug.assert(port < port_count); + const v: u32 = @intFromBool(on); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl10.modify(.{if (port == 0) i2c0_clk_en.is(v) else i2c1_clk_en.is(v)}); +} + +// -------------------------------------------------------------------------------------- timing + +/// Everything the bus timing registers need, in source-clock cycles, as ESP-IDF computes it. +/// +/// The field widths are ESP-IDF's: `i2c_hal_clk_config_t` is nine `uint16_t` +/// (`hal/i2c_types.h:46-56`). That matters at the extremes - a value that would exceed 65535 wraps +/// there too - and it is why this is `u16` rather than `u32`. +pub const Timing = struct { + /// Controller clock divider, as a *count*: the register takes this minus one. + clkm_div: u16, + scl_low: u16, + scl_high: u16, + scl_wait_high: u16, + sda_hold: u16, + sda_sample: u16, + /// Both the start-condition and the stop-condition setup time. + setup: u16, + /// Both the start-condition and the stop-condition hold time. + hold: u16, + /// Timeout *exponent*: the bus times out after 2^tout source-clock cycles. + tout: u16, + + /// Reproduce `i2c_ll_master_cal_bus_clk` (`i2c_ll.h:104-128`) exactly. + /// + /// The whole derivation, because every line of it is load-bearing: + /// + /// clkm_div = source / (bus * 1024) + 1 + /// sclk = source / clkm_div + /// half = sclk / bus / 2 + /// + /// The `+ 1` is not rounding, it is a floor: the period registers are nine bits, so `half` must + /// stay under 512, and dividing the source clock until `sclk <= 1024 * bus` is what guarantees + /// it. At 40 MHz that makes `clkm_div` 1 for every bus frequency above 39 kHz and grows it + /// below - 10 kHz gives `clkm_div` 4, `sclk` 10 MHz, `half` 500 - so the divider is not an + /// optional refinement, it is what makes slow buses representable at all. + /// + /// From `half`, in source-clock cycles: + /// + /// scl_low = half + /// scl_wait_high = half/2 - 2 if bus >= 80 kHz, else half/4 + /// scl_high = half - scl_wait_high + /// sda_hold = half/4 + /// sda_sample = half/2 + /// setup = hold = half + /// tout = 32 - clz(5 * half) + 2 + /// + /// `scl_wait_high` is the part of the high period during which the master waits for the slave to + /// release SCL (clock stretching); `scl_high` is the part it drives. They sum to `half`, so the + /// nominal frequency is the same either way, and IDF's own comment (`i2c_ll.h:112-114`) records + /// why the split changes at 80 kHz: below that, too much wait-high measurably *raises* the + /// frequency on real hardware. + /// + /// The `tout` expression is `log2(5 * half) + 2` written with a count-leading-zeros: a timeout + /// of about 20 half-cycles, i.e. ten bit times, rounded up to the next power of two because the + /// register holds an exponent. IDF writes it as + /// `sizeof(half_cycle) * 8 - __builtin_clz(5 * half_cycle) + 2` with `half_cycle` a `uint32_t`, + /// hence the 32 here. + /// + /// Not reproduced: the `HAL_ASSERT` at `i2c_ll.h:126-127` that + /// `scl_wait_high < sda_sample < scl_high`. It holds for every frequency this can be asked for + /// at 40 MHz (checked from 10 kHz to 1 MHz), and an assert that cannot fire is noise; the + /// ordering it protects is a hardware requirement, not something this code can choose. + pub fn calculate(source_hz: u32, bus_hz: u32) Timing { + std.debug.assert(bus_hz > 0); + std.debug.assert(source_hz / 2 > bus_hz); + + const clkm_div: u32 = source_hz / (bus_hz * 1024) + 1; + const sclk_hz: u32 = source_hz / clkm_div; + const half: u32 = sclk_hz / bus_hz / 2; + + const wait_high: u32 = if (bus_hz >= 80_000) half / 2 - 2 else half / 4; + return .{ + .clkm_div = @truncate(clkm_div), + .scl_low = @truncate(half), + .scl_wait_high = @truncate(wait_high), + .scl_high = @truncate(half - wait_high), + .sda_hold = @truncate(half / 4), + .sda_sample = @truncate(half / 2), + .setup = @truncate(half), + .hold = @truncate(half), + // @clz(0) is 32 in Zig where __builtin_clz(0) is undefined in C, so this differs from + // IDF only for half == 0, which the assert above rules out. + .tout = @truncate(32 - @clz(5 * half) + 2), + }; + } +}; + +/// Write a computed `Timing` to the peripheral's ten timing registers and the controller-clock +/// divider - `i2c_ll_master_set_bus_timing` (`i2c_ll.h:190-220`). +/// +/// **Which values are written minus one and which are not is the substance of this function.** +/// Eight of the ten are `value - 1`, because the hardware counts from zero. `scl_high_period` and +/// `scl_wait_high_period` are written as-is, and that asymmetry is deliberate: IDF's comment +/// (`i2c_ll.h:201-205`) says the Technical Reference Manual asks for minus one on those two as well, +/// and that following it measurably produces an SCL a little *faster* than asked for, so they do not +/// subtract. A port that "fixes" this by making all ten consistent gets a bus that is out of spec at +/// the top end and passes every test that does not include an oscilloscope. +/// +/// Subtractions are done in `u32` with wrapping and truncated by the field write, which is what the +/// C does for a `uint16_t` of 0 as well - it is unreachable here anyway, since `calculate` asserts +/// `half >= 1`. +pub fn applyTiming(port: u8, t: Timing) void { + std.debug.assert(port < port_count); + setClockDivider(port, t.clkm_div); + + scl_low_period.at(port).modify(.{scl_low_period_f.is(@as(u32, t.scl_low) -% 1)}); + // One store where IDF does two read-modify-writes of the same register (`i2c_ll.h:207-208`). + // Same final word; a write-trace comparison sees the difference, a state comparison does not. + scl_high_period.at(port).modify(.{ + scl_high_period_f.is(t.scl_high), + scl_wait_high_period_f.is(t.scl_wait_high), + }); + sda_hold.at(port).modify(.{sda_hold_time.is(@as(u32, t.sda_hold) -% 1)}); + sda_sample.at(port).modify(.{sda_sample_time.is(@as(u32, t.sda_sample) -% 1)}); + scl_rstart_setup.at(port).modify(.{scl_rstart_setup_time.is(@as(u32, t.setup) -% 1)}); + scl_stop_setup.at(port).modify(.{scl_stop_setup_time.is(@as(u32, t.setup) -% 1)}); + scl_start_hold.at(port).modify(.{scl_start_hold_time.is(@as(u32, t.hold) -% 1)}); + scl_stop_hold.at(port).modify(.{scl_stop_hold_time.is(@as(u32, t.hold) -% 1)}); + to.at(port).modify(.{ time_out_value.is(t.tout), time_out_en.is(1) }); +} + +/// Compute and apply the timing for a target SCL frequency. The whole point of the file. +/// +/// Does **not** commit: call `commitConfig` when the rest of the configuration is in place. That is +/// ESP-IDF's division too - `_i2c_hal_set_bus_timing` (`i2c_hal.c:27-32`) is calculate-then-write, +/// and the driver commits separately. +pub fn setBusTiming(port: u8, source_hz: u32, bus_hz: u32) void { + applyTiming(port, Timing.calculate(source_hz, bus_hz)); +} + +/// The controller clock divider: register field is the divider *minus one*, with the fractional +/// numerator and denominator zeroed because ESP-IDF does not use them +/// (`i2c_ll.h:193-199`, `i2c_ll.h:229-239`). +pub fn setClockDivider(port: u8, clkm_div: u16) void { + std.debug.assert(port < port_count); + const num: u32 = @as(u32, clkm_div) -% 1; + const guard = clkrst.maskInterrupts(); + defer guard.release(); + if (port == 0) { + peri_clk_ctrl10.modify(.{ + i2c0_div_num.is(num), + i2c0_div_numerator.is(0), + i2c0_div_denominator.is(0), + }); + } else { + peri_clk_ctrl11.modify(.{ + i2c1_div_num.is(num), + i2c1_div_numerator.is(0), + i2c1_div_denominator.is(0), + }); + } +} + +// The three narrow timing setters, for tuning one condition without recomputing the whole set - a +// slow slave that needs a longer SDA hold, say. +// +// **These do not use the same convention as `applyTiming`, and that is ESP-IDF's inconsistency, not +// a transcription error.** `i2c_ll_master_set_start_timing` writes `scl_rstart_setup = setup` but +// `scl_start_hold = hold - 1` (`i2c_ll.h:452-456`); `i2c_ll_master_set_stop_timing` writes both as +// given (`i2c_ll.h:467-471`); `i2c_ll_set_sda_timing` writes both as given (`i2c_ll.h:482-486`). +// `i2c_ll_master_set_bus_timing`, meanwhile, subtracts one from all six of those +// (`i2c_ll.h:210-217`). The reconciliation is that `cal_bus_clk` produces *cycle counts* and these +// setters take *register values*, with the single exception of `start_hold` - and IDF's own getters +// agree: `i2c_ll_get_start_timing` adds one back to the hold and not to the setup +// (`i2c_ll.h:644-648`), while `i2c_ll_get_stop_timing` adds nothing (`i2c_ll.h:659-663`). Anything +// tidier here would be a different peripheral configuration from the one IDF produces. + +pub fn setStartTiming(port: u8, setup: u32, hold: u32) void { + std.debug.assert(port < port_count); + scl_rstart_setup.at(port).modify(.{scl_rstart_setup_time.is(setup)}); + scl_start_hold.at(port).modify(.{scl_start_hold_time.is(hold -% 1)}); +} + +pub fn setStopTiming(port: u8, setup: u32, hold: u32) void { + std.debug.assert(port < port_count); + scl_stop_setup.at(port).modify(.{scl_stop_setup_time.is(setup)}); + scl_stop_hold.at(port).modify(.{scl_stop_hold_time.is(hold)}); +} + +pub fn setSdaTiming(port: u8, sample: u32, hold: u32) void { + std.debug.assert(port < port_count); + sda_hold.at(port).modify(.{sda_hold_time.is(hold)}); + sda_sample.at(port).modify(.{sda_sample_time.is(sample)}); +} + +/// Timeout exponent for a wanted timeout in microseconds - +/// `i2c_ll_calculate_timeout_us_to_reg_val` (`i2c_ll.h:1060-1065`). +/// +/// `32 - clz(cycles_per_us * timeout_us)` is `log2` rounded *up*, which is the only sensible +/// direction for a bus timeout. IDF's own default for the SCL timeout is 2000 us +/// (`i2c_ll.h:88`). +pub fn timeoutExponent(source_hz: u32, timeout_us: u32) u32 { + const cycles_per_us = source_hz / 1_000_000; + return 32 - @clz(cycles_per_us * timeout_us); +} + +/// Set just the timeout exponent, leaving the enable bit alone - `i2c_ll_set_tout` +/// (`i2c_ll.h:358-361`). The field is five bits: 2^31 source cycles is the longest expressible +/// timeout, which at 40 MHz is 54 seconds. +pub fn setTimeout(port: u8, exponent: u32) void { + std.debug.assert(port < port_count); + to.at(port).modify(.{time_out_value.is(exponent)}); +} + +pub fn setTimeoutEnabled(port: u8, on: bool) void { + std.debug.assert(port < port_count); + to.at(port).modify(.{time_out_en.is(@intFromBool(on))}); +} + +/// Glitch filter: pulses shorter than `cycles` source-clock cycles are ignored on both SDA and SCL. +/// `cycles == 0` disables both filters - `i2c_ll_master_set_filter` (`i2c_ll.h:753-764`). +/// +/// Note what "disable" means here: the two enable bits default to 1 with thresholds of 0, so the +/// reset state is "filtering enabled, filtering nothing", and disabling is not the same word as +/// enabling with a threshold of 0. Passing 0 therefore leaves the thresholds untouched, exactly as +/// IDF does, rather than zeroing them - a difference the register comparison would catch. +pub fn setFilter(port: u8, cycles: u4) void { + std.debug.assert(port < port_count); + const r = filter_cfg.at(port); + if (cycles > 0) { + r.modify(.{ + scl_filter_thres.is(cycles), + sda_filter_thres.is(cycles), + scl_filter_en.is(1), + sda_filter_en.is(1), + }); + } else { + r.modify(.{ scl_filter_en.is(0), sda_filter_en.is(0) }); + } +} + +// ----------------------------------------------------------------------------------- bring-up + +/// Put a port into master mode with the defaults ESP-IDF's `i2c_hal_master_init` establishes +/// (`i2c_hal.c:39-50`), in the same order. +/// +/// The four control bits are one store where IDF does five separate read-modify-writes of the same +/// register; the resulting word is identical. Each one matters: +/// +/// * `ms_mode = 1` - master. +/// * `sda_force_out = scl_force_out = 0` - open drain. The names are inverted: +/// `i2c_ll_enable_pins_open_drain` writes `!enable_od` (`i2c_ll.h:971-975`), so *zero* is +/// open-drain and one is push-pull. Push-pull on a shared bus is a short circuit the moment two +/// devices disagree, so this is the bit that must not be got backwards. +/// * `arbitration_en = 0` - IDF's master init disables arbitration, which defaults to 1. With a +/// single master there is nothing to arbitrate, and a false arbitration-lost abort on a noisy +/// line is worse than none. +/// * `rx_full_ack_level = 0` - ACK, not NACK, when the RX FIFO hits its threshold. +/// * `tx_lsb_first = rx_lsb_first = 0` - MSB first, which is what I2C is. +/// +/// Then both FIFOs are reset, as IDF does, so the block starts with empty FIFOs whatever the +/// previous user left behind. +pub fn initMaster(port: u8) void { + std.debug.assert(port < port_count); + ctr.at(port).modify(.{ + ms_mode.is(1), + sda_force_out.is(0), + scl_force_out.is(0), + arbitration_en.is(0), + rx_full_ack_level.is(0), + tx_lsb_first.is(0), + rx_lsb_first.is(0), + }); + resetTxFifo(port); + resetRxFifo(port); +} + +/// Latch the configuration into the state machine. Write-to-trigger, self-clearing, and required: +/// see note 2 in this file's header. `i2c_ll_update` (`i2c_ll.h:137-141`). +pub inline fn commitConfig(port: u8) void { + ctr.at(port).modify(.{conf_upgate.is(1)}); +} + +/// Reset the master state machine without touching its configuration. Self-clearing in hardware - +/// IDF writes 1 and never writes 0 (`i2c_ll.h:785-789`, "fsm_rst is a self cleared bit"). For a +/// master that has hung mid-transaction; the bus itself may still need `clearBus`. +pub inline fn resetFsm(port: u8) void { + ctr.at(port).modify(.{fsm_rst.is(1)}); +} + +/// Drive up to `pulses` SCL clocks to free a slave that is holding SDA low, then a STOP - +/// `i2c_ll_master_clr_bus` (`i2c_ll.h:803-810`). Nine pulses is IDF's default +/// (`I2C_LL_RESET_SLV_SCL_PULSE_NUM_DEFAULT`, `i2c_ll.h:87`): enough for any slave to finish the +/// byte it is stuck in and see a NACK. +/// +/// The enable bit is cleared *by hardware* when the pulses have been sent, so completion is polled +/// through `isBusClearDone`, and `commitConfig` is needed both to start it and, per IDF's comment, +/// to resynchronise afterwards. Only meaningful with SCL and SDA actually routed to pads. +pub fn clearBus(port: u8, pulses: u5) void { + std.debug.assert(port < port_count); + scl_sp_conf.at(port).modify(.{ scl_rst_slv_num.is(pulses), scl_rst_slv_en.is(1) }); + commitConfig(port); +} + +pub inline fn isBusClearDone(port: u8) bool { + return scl_sp_conf.at(port).get(scl_rst_slv_en) == 0; +} + +/// Open-drain or push-pull SCL and SDA, at the peripheral end. +/// +/// **The register fields are the inverse of this argument.** `i2c_ll_enable_pins_open_drain` writes +/// `sda_force_out = scl_force_out = !enable_od` (`i2c_ll.h:971-975`), so a zero in either field is +/// what makes that line release instead of driving high. `initMaster` already establishes +/// open-drain; this exists to be able to change it, and to have the polarity checked against IDF's +/// on its own rather than only as part of a seven-field store. +/// +/// This is the *peripheral's* driver behaviour. The pad also has an open-drain bit of its own in the +/// GPIO block (`gpio.setOpenDrain`), and a real bus needs both: the pad hardware must not drive +/// high, and the peripheral must not ask it to. +pub fn setPinsOpenDrain(port: u8, open_drain: bool) void { + std.debug.assert(port < port_count); + const v: u32 = @intFromBool(!open_drain); + ctr.at(port).modify(.{ sda_force_out.is(v), scl_force_out.is(v) }); +} + +// --------------------------------------------------------------------------------------- FIFOs + +/// FIFO or RAM access. FIFO mode is `nonfifo_en = 0`, i.e. the field is the inverse of the name of +/// this function - `i2c_ll_enable_fifo_mode` (`i2c_ll.h:345-348`). +pub fn setFifoMode(port: u8, fifo: bool) void { + std.debug.assert(port < port_count); + fifo_conf.at(port).modify(.{nonfifo_en.is(@intFromBool(!fifo))}); +} + +/// Hold the TX FIFO in reset, then release it. Two stores, because the bit is plain R/W and not +/// self-clearing: writing only the 1 leaves the FIFO permanently reset and every subsequent +/// transmission silently empty (`i2c_ll.h:248-253`). +pub fn resetTxFifo(port: u8) void { + std.debug.assert(port < port_count); + const r = fifo_conf.at(port); + r.modify(.{tx_fifo_rst.is(1)}); + r.modify(.{tx_fifo_rst.is(0)}); +} + +pub fn resetRxFifo(port: u8) void { + std.debug.assert(port < port_count); + const r = fifo_conf.at(port); + r.modify(.{rx_fifo_rst.is(1)}); + r.modify(.{rx_fifo_rst.is(0)}); +} + +/// FIFO watermark thresholds, and the two side effects ESP-IDF attaches to setting them. +/// +/// `fifo_prt_en` gates the watermark interrupts *and* the overflow/underflow protection +/// (`i2c_reg.h:449-459`), and IDF sets it in both threshold setters +/// (`i2c_ll.h:496-500` and `i2c_ll.h:510-515`), so it is set here rather than left to the caller. +/// +/// The other side effect is less obvious and is copied deliberately: IDF's +/// `i2c_ll_set_rxfifo_full_thr` also writes `ctr.rx_full_ack_level = 0`, in a different register. +/// That is coherent rather than sloppy - an RX threshold means "ACK up to here", and a master that +/// NACKed at the threshold would end the transfer instead of pausing it - but it means this +/// operation touches two registers, and after a peripheral reset (where `rx_full_ack_level` defaults +/// to 1) leaving it out is an observable difference rather than a stylistic one. +pub fn setFifoThresholds(port: u8, tx_empty: u5, rx_full: u5) void { + std.debug.assert(port < port_count); + fifo_conf.at(port).modify(.{ + fifo_prt_en.is(1), + txfifo_wm_thrhd.is(tx_empty), + rxfifo_wm_thrhd.is(rx_full), + }); + ctr.at(port).modify(.{rx_full_ack_level.is(0)}); +} + +// ------------------------------------------------------------------------------- Data register +// +// **Reading `I2C_DATA_REG` pops the RX FIFO.** The register header does not say so - it annotates +// the single field `I2C_FIFO_RDATA` as `HRO` and describes the register as "Rx FIFO read data" +// (`i2c_reg.h:464-474`) - but ESP-IDF's LL settles it: `i2c_ll_read_rxfifo` reads *the same address* +// `len` times into successive bytes of a buffer (`i2c_ll.h:691-697`), which can only produce +// distinct bytes if each read advances the FIFO. The write direction is the same address for the +// other FIFO: `i2c_ll_write_txfifo` stores `len` bytes to `hw->data.val` (`i2c_ll.h:674-680`). One +// address, two FIFOs, both with side effects - the same shape as `UART_FIFO_REG`, and the reason +// this offset is in the differential harness's `no_read` list. + +/// Push bytes into the TX FIFO. In FIFO mode each store is one byte into the FIFO regardless of the +/// width of the access; the FIFO is `fifo_len` deep and there is no flow control here, so the caller +/// must not exceed `txSpace`. +pub fn writeTxFifo(port: u8, bytes: []const u8) void { + std.debug.assert(port < port_count); + std.debug.assert(bytes.len <= fifo_len); + const r = data.at(port); + for (bytes) |b| r.writeRaw(b); +} + +/// Pop bytes out of the RX FIFO. Destructive by construction - see above. +pub fn readRxFifo(port: u8, out: []u8) void { + std.debug.assert(port < port_count); + const r = data.at(port); + for (out) |*b| b.* = @truncate(r.raw()); +} + +/// Bytes waiting in the RX FIFO. +pub inline fn rxCount(port: u8) u32 { + return sr.at(port).get(rxfifo_cnt); +} + +/// Bytes queued in the TX FIFO. +pub inline fn txCount(port: u8) u32 { + return sr.at(port).get(txfifo_cnt); +} + +/// Room left in the TX FIFO, saturating at 0 the way `i2c_ll_get_txfifo_len` does +/// (`i2c_ll.h:604-608`) - the counter can read `fifo_len` and the subtraction must not wrap. +pub inline fn txSpace(port: u8) u32 { + const used = txCount(port); + return if (used >= fifo_len) 0 else fifo_len - used; +} + +pub inline fn isBusBusy(port: u8) bool { + return sr.at(port).get(bus_busy) == 1; +} + +// -------------------------------------------------------------------------------- command list +// +// A transaction is up to eight commands written into I2C_COMD0..7 and then triggered as a unit. The +// register header exposes each slot as a single 14-bit field `I2C_COMMANDn` plus a `_DONE` bit at 31 +// and stops there: the sub-fields exist only in `i2c_ll_hw_cmd_t` (`i2c_ll.h:41-52`). So this is one +// of the few places where the field geometry cannot come from a macro pair, and the comptime check +// below is what keeps that honest - the five sub-fields must tile exactly the bits the header calls +// I2C_COMMANDn. + +const cmd_byte_num = Field.of(0, 0xff); +const cmd_ack_en = Field.bit(8); +const cmd_ack_exp = Field.bit(9); +const cmd_ack_val = Field.bit(10); +const cmd_op_code = Field.of(11, 0x7); +const cmd_done = Field.of(regs.I2C_COMMAND0_DONE_S, regs.I2C_COMMAND0_DONE_V); + +comptime { + const command_field = Field.of(regs.I2C_COMMAND0_S, regs.I2C_COMMAND0_V); + const tiled = cmd_byte_num.mask() | cmd_ack_en.mask() | cmd_ack_exp.mask() | + cmd_ack_val.mask() | cmd_op_code.mask(); + if (tiled != command_field.mask()) @compileError( + "the command sub-fields from i2c_ll.h do not tile I2C_COMMAND0 - one of the two headers moved", + ); + if (cmd_done.mask() & command_field.mask() != 0) @compileError("command done bit overlaps the command"); +} + +/// Opcodes, from `i2c_ll.h:55-59`. **Not** the numbers this chip's own register header describes - +/// see note 3 in the file header. +pub const Op = enum(u3) { + write = 1, + stop = 2, + read = 3, + /// Hand the command list back to software with the bus still held, so the next chunk can be + /// loaded. This is how a transfer longer than eight commands or 32 bytes is done without DMA. + end = 4, + /// START, and equally a repeated START. + restart = 6, +}; + +/// One command slot as a value rather than a raw word. +/// +/// The three ACK fields only mean something for one direction each, which is why they are separate +/// rather than one "ack" number: +/// +/// * `ack_check` (WRITE) - compare the ACK bit the slave returns against `ack_expected` and abort +/// the list if it differs. This is what turns a missing device into a NACK error instead of a +/// transfer into the void. +/// * `ack_value` (READ) - the ACK bit this master sends after each byte it reads. Zero (ACK) for +/// every byte but the last, one (NACK) for the last, which is how a slave is told to stop +/// driving the bus. +pub const Command = struct { + op: Op, + /// Bytes to move. Only WRITE and READ use it; a READ of n bytes is one command, not n. + bytes: u8 = 0, + ack_check: bool = false, + ack_expected: u1 = 0, + ack_value: u1 = 0, + + pub inline fn encode(self: Command) u32 { + return (@as(u32, self.bytes) << cmd_byte_num.shift) | + (@as(u32, @intFromBool(self.ack_check)) << cmd_ack_en.shift) | + (@as(u32, self.ack_expected) << cmd_ack_exp.shift) | + (@as(u32, self.ack_value) << cmd_ack_val.shift) | + (@as(u32, @intFromEnum(self.op)) << cmd_op_code.shift); + } +}; + +/// One command slot. The slot stride is checked against the header's own COMD1 macro rather than +/// assumed to be 4. +inline fn cmdReg(port: u8, slot: u8) Reg { + std.debug.assert(slot < cmd_slots); + const stride = comptime mmio.addr(regs.I2C_COMD1_REG(0)) - mmio.addr(regs.I2C_COMD0_REG(0)); + comptime { + // ... and the array is contiguous all the way to the last slot. + if (mmio.addr(regs.I2C_COMD7_REG(0)) != mmio.addr(regs.I2C_COMD0_REG(0)) + stride * 7) + @compileError("the command registers are not a contiguous array of 8"); + } + return Reg.atAddress(comd0.at(port).address + stride * slot); +} + +/// Write a command into a slot. A whole-word store, as IDF's `i2c_ll_master_write_cmd_reg` does +/// (`i2c_ll.h:437-441`): it is the one register here where establishing the entire word is right, +/// because the `done` bit must go back to 0 for the slot to be waited on again. +pub fn writeCommand(port: u8, slot: u8, cmd: Command) void { + std.debug.assert(port < port_count); + cmdReg(port, slot).writeRaw(cmd.encode()); +} + +/// Load a whole command list, in order. Any slot the list does not reach keeps whatever it held - +/// which is harmless, because the sequencer stops at the STOP or END that the list must contain. +pub fn writeCommands(port: u8, cmds: []const Command) void { + std.debug.assert(cmds.len <= cmd_slots); + for (cmds, 0..) |c, i| writeCommand(port, @intCast(i), c); +} + +/// Whether the sequencer has finished a slot. Set by hardware (`R/W/SS`), cleared by writing the +/// slot again. `i2c_ll_master_is_cmd_done` (`i2c_ll.h:1047-1051`). +pub inline fn isCommandDone(port: u8, slot: u8) bool { + return cmdReg(port, slot).get(cmd_done) == 1; +} + +// --------------------------------------------------------------------------------- transactions + +// The master event bits, in I2C_INT_RAW/I2C_INT_ST/I2C_INT_CLR - the same bit numbers in all three +// (`i2c_ll.h:61-70`). Reading INT_RAW is safe: the bits are `R/SS/WTC`, set by hardware and cleared +// only by writing a 1 to the same position in INT_CLR, so polling does not consume them. Writing +// INT_CLR is the one place in this file that must be `writeRaw` rather than `modify`. +const int_trans_complete = Field.of(regs.I2C_TRANS_COMPLETE_INT_RAW_S, regs.I2C_TRANS_COMPLETE_INT_RAW_V); +const int_end_detect = Field.of(regs.I2C_END_DETECT_INT_RAW_S, regs.I2C_END_DETECT_INT_RAW_V); +const int_nack = Field.of(regs.I2C_NACK_INT_RAW_S, regs.I2C_NACK_INT_RAW_V); +const int_arbitration_lost = Field.of(regs.I2C_ARBITRATION_LOST_INT_RAW_S, regs.I2C_ARBITRATION_LOST_INT_RAW_V); +const int_time_out = Field.of(regs.I2C_TIME_OUT_INT_RAW_S, regs.I2C_TIME_OUT_INT_RAW_V); +const int_scl_st_to = Field.of(regs.I2C_SCL_ST_TO_INT_RAW_S, regs.I2C_SCL_ST_TO_INT_RAW_V); +const int_scl_main_st_to = Field.of(regs.I2C_SCL_MAIN_ST_TO_INT_RAW_S, regs.I2C_SCL_MAIN_ST_TO_INT_RAW_V); + +/// The mask ESP-IDF uses for "all interrupts" - `I2C_LL_INTR_MASK`, `i2c_ll.h:1097`. +/// +/// It is 14 bits, and this block has 19 (`I2C_SLAVE_ADDR_UNMATCH_INT` is bit 18). The five it leaves +/// out are slave-mode and general-call events, which is presumably why IDF's mask stops where it +/// does; the value is IDF's rather than a recount so that clearing "everything" means the same thing +/// on both sides of the differential. +pub const all_interrupts: u32 = 0x3fff; + +/// Clear interrupt flags. Write-1-to-clear, so this is a raw store of a mask and never a +/// read-modify-write: reading INT_RAW and writing it back would clear whatever had arrived in +/// between and nothing else. +pub inline fn clearInterrupts(port: u8, mask: u32) void { + int_clr.at(port).writeRaw(mask); +} + +/// Mask every interrupt at the peripheral. This HAL polls; nothing here reaches the CLIC. +/// +/// A whole-word zero rather than IDF's `int_ena &= ~mask` (`i2c_ll.h:305-309`), so it also covers +/// the five slave-mode bits outside `all_interrupts`. Reaching the same word from a block whose +/// `int_ena` reset value is 0 either way, which is why the differential case for it agrees. +pub inline fn disableInterrupts(port: u8) void { + int_ena.at(port).writeRaw(0); +} + +/// How a triggered command list ended. +pub const Outcome = enum { + /// The list ran to its STOP. + complete, + /// The list hit an END opcode: the bus is still held and the next chunk can be loaded. + end_detect, + /// A slave did not acknowledge. The usual meaning is "nothing at that address". + nack, + /// Another master won the bus. Only possible with `arbitration_en` set, which `initMaster` + /// clears. + arbitration_lost, + /// SCL was held low past the configured timeout - `I2C_TO_REG`. Almost always a slave holding + /// the clock, or no pull-up on the line at all. + timeout, + /// The SCL state machine stalled: `scl_st_to` or `scl_main_st_to`. IDF's driver treats this as + /// the signal that a bus deadlock may have happened and `clearBus` is worth trying + /// (`i2c_ll.h:795`). + stalled, + /// Nothing had happened yet. + pending, +}; + +/// Trigger the loaded command list. Write-to-trigger; the bit reads back 0, so this leaves no trace +/// in a register snapshot. `i2c_ll_start_trans` (`i2c_ll.h:629-633`). +pub inline fn startTransaction(port: u8) void { + ctr.at(port).modify(.{trans_start.is(1)}); +} + +/// Read the outcome so far from one load of INT_RAW. +/// +/// Errors are reported ahead of completion, and in the order they matter: an arbitration loss or a +/// NACK can be raised in the same word as `trans_complete`, and calling that transaction complete +/// is how a driver comes to believe a device answered when it did not. +pub fn outcome(port: u8) Outcome { + const raw = int_raw.at(port).raw(); + if (raw & int_arbitration_lost.mask() != 0) return .arbitration_lost; + if (raw & int_nack.mask() != 0) return .nack; + if (raw & int_time_out.mask() != 0) return .timeout; + if (raw & (int_scl_st_to.mask() | int_scl_main_st_to.mask()) != 0) return .stalled; + if (raw & int_trans_complete.mask() != 0) return .complete; + if (raw & int_end_detect.mask() != 0) return .end_detect; + return .pending; +} + +/// Spin until the transaction resolves. Returns `.pending` if it never does, rather than hanging: +/// a bus with no pull-up produces exactly that, and it is a fault to report rather than a board to +/// power-cycle. +/// +/// `spins` is a loop count, not a time. At the ~90 MHz this board boots at, a 100 kHz transfer of a +/// few bytes needs on the order of 10^4 iterations of this loop; the default of 200,000 leaves an +/// order of magnitude of headroom and still returns in well under a second. +pub fn waitTransaction(port: u8, spins: u32) Outcome { + var n: u32 = 0; + while (n < spins) : (n += 1) { + const o = outcome(port); + if (o != .pending) return o; + } + return .pending; +} + +/// The status register's own error bits, which are not the interrupt flags: `resp_rec` is the last +/// ACK level *received* and `arb_lost` is the state machine's own latch. Both are read-only and +/// survive an interrupt clear, so they are what to look at when diagnosing a transfer after the fact. +pub const Status = struct { + /// The ACK bit the slave last returned: 0 = ACK, 1 = NACK. + last_ack: u1, + arbitration_lost: bool, + bus_busy: bool, + rx_bytes: u32, + tx_bytes: u32, +}; + +pub fn status(port: u8) Status { + const raw = sr.at(port).raw(); + return .{ + .last_ack = @intCast((raw >> resp_rec.shift) & 1), + .arbitration_lost = raw & arb_lost.mask() != 0, + .bus_busy = raw & bus_busy.mask() != 0, + .rx_bytes = (raw >> rxfifo_cnt.shift) & rxfifo_cnt.unshiftedMask(), + .tx_bytes = (raw >> txfifo_cnt.shift) & txfifo_cnt.unshiftedMask(), + }; +} + +// ------------------------------------------------------------------------------------ the pads +// +// I2C is a two-wire open-drain bus and the P4 reaches it only through the GPIO matrix: there is no +// IO MUX function for I2C on any pad, so both signals go out through `matrixOut` and come back in +// through `matrixIn`. Both directions are needed even for a write-only master - the master samples +// SDA to read the slave's ACK, and samples SCL to detect stretching - which is why every pad here +// gets its input buffer enabled as well as its driver. + +/// The GPIO matrix signal indices for a port, from ESP-IDF's own signal map +/// (`gpio_sig_map.h:141-148`) via `i2c_periph.c`. On this chip a signal's input and output index +/// happen to be the same number, which is not true on every part and is not something to rely on. +pub fn sclSignal(port: u8) u32 { + return switch (port) { + 0 => regs.I2C0_SCL_PAD_OUT_IDX, + else => regs.I2C1_SCL_PAD_OUT_IDX, + }; +} + +pub fn sdaSignal(port: u8) u32 { + return switch (port) { + 0 => regs.I2C0_SDA_PAD_OUT_IDX, + else => regs.I2C1_SDA_PAD_OUT_IDX, + }; +} + +/// Route SCL and SDA to two pads, open-drain, following `i2c_common_set_pins` +/// (`esp_driver_i2c/i2c_common.c:318-345`) step for step. +/// +/// **The internal pull-ups are not enough for a real bus.** They are on the order of 45 kOhm, which +/// with a few tens of picofarads of trace and device capacitance gives a rise time far past the +/// 1 us that 100 kHz I2C allows. ESP-IDF says the same thing in its own driver documentation and +/// enables them anyway as a convenience for a single device on a short wire. A bus that is expected +/// to work needs external resistors - 4.7 kOhm to 3.3 V is the usual choice at 100 kHz, 2.2 kOhm at +/// 400 kHz - and then `internal_pullups` should be false, because two resistors in parallel is not +/// what either calculation assumed. +/// +/// The order matters in one place: the pad is driven high *before* its output is enabled, so +/// enabling the driver cannot pull the bus low for the few cycles before the peripheral takes over. +/// A low SCL glitch is a clock edge to every device on the bus. +pub fn configurePins(port: u8, scl_pin: u8, sda_pin: u8, opts: struct { + internal_pullups: bool = false, +}) void { + std.debug.assert(port < port_count); + for ([_]struct { pin: u8, signal: u32 }{ + .{ .pin = scl_pin, .signal = sclSignal(port) }, + .{ .pin = sda_pin, .signal = sdaSignal(port) }, + }) |wire| { + gpio.setHigh(wire.pin); + gpio.setInputEnable(wire.pin, true); + gpio.setOpenDrain(wire.pin, true); + gpio.setPull(wire.pin, if (opts.internal_pullups) .up else .none); + gpio.matrixOut(wire.pin, wire.signal); + gpio.matrixIn(wire.pin, wire.signal); + } +} + +// ------------------------------------------------------------------------------- transfers + +/// Bring a port up as a master on a given bus frequency, in the order the hardware requires: +/// clocks, then reset, then configuration, then commit. +/// +/// Reset before configure, because a reset drops everything configured before it. `clkrst.init` +/// does the gate-then-reset pair; the controller clock is separate and enabled after, since it only +/// feeds the state machine. +pub fn init(port: u8, opts: struct { + source: Source = .xtal, + source_hz: u32 = xtal_hz, + bus_hz: u32 = 100_000, + /// Glitch filter width in source-clock cycles. ESP-IDF's driver default is 7. + filter_cycles: u4 = 7, +}) void { + std.debug.assert(port < port_count); + switch (port) { + 0 => clkrst.init(.i2c0), + else => clkrst.init(.i2c1), + } + setControllerClockEnabled(port, true); + setSource(port, opts.source); + + initMaster(port); + setFifoMode(port, true); + disableInterrupts(port); + clearInterrupts(port, all_interrupts); + setBusTiming(port, opts.source_hz, opts.bus_hz); + setFilter(port, opts.filter_cycles); + commitConfig(port); +} + +/// Default spin budget for `write`/`read`. See `waitTransaction`. +pub const default_spins: u32 = 200_000; + +/// Write `bytes` to a 7-bit address as one command list. +/// +/// RSTART | WRITE (1 + len bytes, ack checked) | STOP +/// +/// The address byte goes in the TX FIFO ahead of the data and is counted in the WRITE command's byte +/// count: to the sequencer the address is just the first byte written after a START. `ack_check` is +/// on, so a missing device comes back as `.nack` rather than as a successful write into nothing. +/// +/// One command list, one FIFO load: at most `fifo_len - 1` = 31 data bytes. Longer transfers need +/// the END-and-continue loop that ESP-IDF's driver runs from its interrupt handler, which is out of +/// scope here - hence the assert rather than a partial write. +pub fn write(port: u8, address: u7, bytes: []const u8, spins: u32) Outcome { + std.debug.assert(bytes.len < fifo_len); + resetTxFifo(port); + resetRxFifo(port); + clearInterrupts(port, all_interrupts); + + writeTxFifo(port, &[_]u8{@as(u8, address) << 1}); + writeTxFifo(port, bytes); + + writeCommands(port, &.{ + .{ .op = .restart }, + .{ .op = .write, .bytes = @intCast(bytes.len + 1), .ack_check = true }, + .{ .op = .stop }, + }); + commitConfig(port); + startTransaction(port); + return waitTransaction(port, spins); +} + +/// Read into `out` from a 7-bit address as one command list. +/// +/// RSTART | WRITE 1 (address|read, ack checked) | READ n-1 sending ACK | READ 1 sending NACK | STOP +/// +/// The last byte is a separate command because its ACK bit differs: a master that ACKs the final +/// byte tells the slave to keep going, and the slave then holds SDA for a byte that will never be +/// clocked out. That is the classic I2C read bug, and it is a *command list* bug - which is why the +/// split is here rather than being something the caller can get wrong. +/// +/// Reads of one byte collapse to a single NACKed READ, so the list is four commands instead of five. +pub fn read(port: u8, address: u7, out: []u8, spins: u32) Outcome { + std.debug.assert(out.len > 0); + std.debug.assert(out.len <= fifo_len); + resetTxFifo(port); + resetRxFifo(port); + clearInterrupts(port, all_interrupts); + + writeTxFifo(port, &[_]u8{(@as(u8, address) << 1) | 1}); + + writeCommand(port, 0, .{ .op = .restart }); + writeCommand(port, 1, .{ .op = .write, .bytes = 1, .ack_check = true }); + var slot: u8 = 2; + if (out.len > 1) { + writeCommand(port, slot, .{ .op = .read, .bytes = @intCast(out.len - 1), .ack_value = 0 }); + slot += 1; + } + writeCommand(port, slot, .{ .op = .read, .bytes = 1, .ack_value = 1 }); + writeCommand(port, slot + 1, .{ .op = .stop }); + + commitConfig(port); + startTransaction(port); + const result = waitTransaction(port, spins); + if (result == .complete) readRxFifo(port, out); + return result; +} + +test "the timing arithmetic reproduces ESP-IDF's, including where it looks wrong" { + // 100 kHz on a 40 MHz XTAL: the case every I2C device supports, worked through by hand from + // i2c_ll.h:104-128. clkm_div = 40e6/(100e3*1024) + 1 = 0 + 1 = 1, so sclk stays 40 MHz and + // half = 40e6/100e3/2 = 200. + const t100 = Timing.calculate(40_000_000, 100_000); + try std.testing.expectEqual(@as(u16, 1), t100.clkm_div); + try std.testing.expectEqual(@as(u16, 200), t100.scl_low); + try std.testing.expectEqual(@as(u16, 98), t100.scl_wait_high); // half/2 - 2 + try std.testing.expectEqual(@as(u16, 102), t100.scl_high); // half - wait_high + try std.testing.expectEqual(@as(u16, 50), t100.sda_hold); + try std.testing.expectEqual(@as(u16, 100), t100.sda_sample); + try std.testing.expectEqual(@as(u16, 200), t100.setup); + try std.testing.expectEqual(@as(u16, 200), t100.hold); + // 5*200 = 1000, which needs 10 bits, so 32 - 22 + 2 = 12: a timeout of 2^12 = 4096 cycles, + // 102 us at 40 MHz, about ten bit times. + try std.testing.expectEqual(@as(u16, 12), t100.tout); + + // 400 kHz: same divider, quarter the half-cycle. + const t400 = Timing.calculate(40_000_000, 400_000); + try std.testing.expectEqual(@as(u16, 1), t400.clkm_div); + try std.testing.expectEqual(@as(u16, 50), t400.scl_low); + try std.testing.expectEqual(@as(u16, 23), t400.scl_wait_high); + try std.testing.expectEqual(@as(u16, 27), t400.scl_high); + try std.testing.expectEqual(@as(u16, 10), t400.tout); + + // 10 kHz: the branch that actually uses the controller-clock divider. 40e6/(10e3*1024) = 3, so + // clkm_div = 4, sclk = 10 MHz and half = 500 - just inside the nine-bit period fields, which is + // what the divider exists to guarantee. + const t10 = Timing.calculate(40_000_000, 10_000); + try std.testing.expectEqual(@as(u16, 4), t10.clkm_div); + try std.testing.expectEqual(@as(u16, 500), t10.scl_low); + // Below 80 kHz the wait-high split changes: half/4 rather than half/2 - 2. + try std.testing.expectEqual(@as(u16, 125), t10.scl_wait_high); + try std.testing.expectEqual(@as(u16, 375), t10.scl_high); + + // The hardware ordering constraint IDF asserts (i2c_ll.h:126-127) across the whole range. + for ([_]u32{ 10_000, 50_000, 100_000, 400_000, 1_000_000 }) |hz| { + const t = Timing.calculate(40_000_000, hz); + try std.testing.expect(t.scl_wait_high < t.sda_sample); + try std.testing.expect(t.sda_sample < t.scl_high); + // Every period register is nine bits wide, and scl_low is written minus one. + try std.testing.expect(t.scl_low - 1 <= 511); + try std.testing.expect(t.scl_wait_high <= 127); // this one is seven + try std.testing.expect(t.tout <= 31); // and the timeout exponent is five + } +} + +test "the timeout exponent rounds up, and where the five-bit field runs out" { + // 2000 us at 40 MHz is 80,000 cycles; 2^17 = 131,072 is the first power of two above it, so + // IDF's documented default SCL timeout comes out as 17 - which fits the five-bit field with + // room to spare. This test exists because the first version of this file asserted the opposite. + try std.testing.expectEqual(@as(u32, 17), timeoutExponent(40_000_000, 2000)); + try std.testing.expect(timeoutExponent(40_000_000, 2000) <= time_out_value.max()); + // The field runs out at 2^31 source cycles, 53.7 seconds at 40 MHz - a timeout no I2C bus has a + // use for, which is why neither IDF nor this file range-checks it. Past that the exponent is + // truncated by the field write rather than rejected, exactly as IDF's bitfield store does. + try std.testing.expectEqual(@as(u32, 32), timeoutExponent(40_000_000, 100_000_000)); + try std.testing.expect(timeoutExponent(40_000_000, 100_000_000) > time_out_value.max()); +} + +test "commands encode to the layout i2c_ll_hw_cmd_t describes" { + // A WRITE of three bytes with ACK checking: byte_num=3, ack_en=1, op_code=1. + try std.testing.expectEqual( + @as(u32, 3) | (1 << 8) | (1 << 11), + (Command{ .op = .write, .bytes = 3, .ack_check = true }).encode(), + ); + // RESTART is opcode 6 on this chip, not 0 - the number the register header's prose still gives. + try std.testing.expectEqual(@as(u32, 6 << 11), (Command{ .op = .restart }).encode()); + // A final READ NACKs: ack_val=1 at bit 10, opcode 3. + try std.testing.expectEqual( + @as(u32, 1) | (1 << 10) | (3 << 11), + (Command{ .op = .read, .bytes = 1, .ack_value = 1 }).encode(), + ); + // STOP is 2 and READ is 3, which is the pair the ESP32-era numbering had the other way around. + try std.testing.expectEqual(@as(u32, 2 << 11), (Command{ .op = .stop }).encode()); +} diff --git a/src/hal/intr.zig b/src/hal/intr.zig new file mode 100644 index 0000000..38f8789 --- /dev/null +++ b/src/hal/intr.zig @@ -0,0 +1,965 @@ +//! The interrupt controller. The ESP32-P4 has a **CLIC**, not a PLIC and not the Xtensa-style +//! fixed matrix of the older parts: `soc_caps.h:191` defines SOC_INT_CLIC_SUPPORTED 1, and +//! `soc/interrupt_reg.h:16` says so in prose. Three consequences shape this file. +//! +//! **1. Two independent stages.** A peripheral source does not have a CPU interrupt number; it has +//! a *mapping register*. The interrupt matrix at DR_REG_INTERRUPT_CORE0_BASE holds one 6-bit word +//! per source, and writing `line + 16` into it points that source at external CLIC line `line`. +//! The `+ 16` is not decoration: the CLIC's first 16 IDs are the RISC-V internal interrupts +//! (software, timer, external), so the 32 lines a driver may use are IDs 16..47. +//! `hal/interrupt_clic_ll.h:35-48` is the matrix write; the `+ RV_EXTERNAL_INT_OFFSET` that turns a +//! line number into a CLIC ID is one level up, at `riscv/interrupt_clic.c:26`. Per-line control - +//! enable, trigger, priority, pending - is the *other* stage, in the CLIC's own register file at +//! DR_REG_CLIC_CTRL_BASE, and it is indexed by CLIC ID, i.e. by `line + 16` again. +//! +//! **2. The threshold is a memory-mapped register on this die, not the `mintthresh` CSR.** This is +//! the single easiest thing to get wrong here, because every RISC-V CLIC document and every +//! ESP32-P4 rev-3 build says `mintthresh` (CSR 0x347). `soc/interrupt_reg.h:28-40` selects +//! `INTTHRESH_STANDARD 0` under CONFIG_ESP32P4_SELECTS_REV_LESS_V3 - the same condition that +//! selects the `register/hw_ver1` headers this project builds against - and +//! `riscv/csr_clic.h:37-47` then leaves MINTTHRESH_CSR *undefined*. The threshold lives in +//! CLIC_INT_THRESH_REG at 0x2080_0008, bits [31:24] (`soc/clic_reg.h:61-67`). Writing CSR 0x347 on +//! this silicon is not an illegal instruction and not an error; it writes a register the interrupt +//! arbiter does not read, so interrupts stay masked and nothing says why. +//! +//! **3. `regs.INTTHRESH_STANDARD` lies, and must not be used.** The register module is +//! `zig translate-c` over the headers with *no* sdkconfig, so CONFIG_ESP32P4_SELECTS_REV_LESS_V3 is +//! absent there and `interrupt_reg.h` takes its `#else` branch: the translated module contains +//! `pub const INTTHRESH_STANDARD = 1`, which is the wrong answer for this die. (The oracle's C side +//! is compiled against `src/oracle/oracle_sdkconfig.h:25`, which does define it, so IDF's own code +//! there takes the correct branch. The two disagree, deliberately, and only the C side is right +//! about this macro.) Nothing in this file reads it. +//! +//! Nothing below has been run on hardware by the author of this file. What is claimed is that the +//! register arithmetic matches ESP-IDF's at the cited lines, and that `src/oracle/intr_cases.zig` +//! compares the two on the die. Taking an actual interrupt is a behavioural property no register +//! comparison can establish; see the note at the foot of that file. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +// ------------------------------------------------------------------------------- geometry + +/// CLIC IDs 0..15 are the RISC-V internal interrupts; a driver cannot have them. IDs 16..47 are the +/// 32 external lines. `riscv/csr_clic.h:28-29` (RV_EXTERNAL_INT_COUNT, RV_EXTERNAL_INT_OFFSET) and +/// `soc/clic_reg.h:14` (CLIC_EXT_INTR_NUM_OFFSET) are three names for these two numbers. +pub const line_count: u32 = 32; +pub const ext_offset: u32 = @intCast(regs.CLIC_EXT_INTR_NUM_OFFSET); +/// 16 internal + 32 external. `hal/interrupt_clic_ll.h:22` RV_TOTAL_INT_COUNT, and the hardware +/// agrees: CLIC_INT_INFO_REG's NUM_INT field reads 48 at reset (`soc/clic_reg.h:54-59`). +pub const total_ids: u32 = 48; + +/// Priority levels. `soc/clic_reg.h:13` NLBITS 3, so 8 levels, held in the *top* 3 bits of the +/// 8-bit CLIC_INT_CTL field. Level 0 is masked by the reset threshold; a usable interrupt wants 1 +/// or more. +pub const NLBITS: u5 = @intCast(regs.NLBITS); +const nlbits_shift: u5 = 8 - NLBITS; +/// The low `8 - NLBITS` bits of a priority/threshold byte are not part of the level and IDF fills +/// them with ones (`riscv/csr_clic.h:59`, NLBITS_TO_BYTE). Reproduced exactly, because the +/// differential compares the whole word. +const nlbits_pad: u32 = (@as(u32, 1) << nlbits_shift) - 1; + +// -------------------------------------------------------------------------- interrupt matrix + +/// Every peripheral interrupt source on this chip, from `soc/interrupts.h` - which opens with +/// "This table is decided by hardware, don't touch this." +/// +/// IDs 0..127 are contiguous and each has a mapping register at `matrix_base + 4*id`: the last of +/// them, `assist_debug` = 127, is INTERRUPT_CORE0_ASSIST_DEBUG_INT_MAP_REG at +0x1FC, which is +/// exactly 4*127. That is the invariant `interrupt_clic_ll.h:46` depends on when it computes the +/// address arithmetically rather than from a table. +/// +/// **The last three exist only on chip revision >= 3.0 and therefore not on this die.** +/// `soc/interrupts.h:155-160` explains the gap: their mapping registers are *not* contiguous with +/// the rest, so IDF gave them IDs 133-135 to make `base + 4*id` land on the right address anyway. +/// The numbering hole at 128..132 is that workaround, not missing hardware. On a pre-v3 part - +/// which is what `regs.ZIG_P4_HW_VER == 1` asserts - routing one of them writes a register that +/// nothing drives. +pub const Source = enum(u8) { + lp_rtc = 0, + lp_wdt = 1, + lp_timer_reg0 = 2, + lp_timer_reg1 = 3, + mb_hp = 4, + mb_lp = 5, + pmu_0 = 6, + pmu_1 = 7, + lp_anaperi = 8, + lp_adc = 9, + lp_gpio = 10, + lp_i2c = 11, + lp_i2s = 12, + lp_spi = 13, + lp_touch = 14, + /// Also spelled ETS_TEMPERATURE_SENSOR_INTR_SOURCE; IDF aliases the two (`interrupts.h:34`). + lp_tsens = 15, + lp_uart = 16, + lp_efuse = 17, + lp_sw = 18, + lp_sysreg = 19, + lp_huk = 20, + sys_icm = 21, + usb_serial_jtag = 22, + sdio_host = 23, + dw_gdma = 24, + spi2 = 25, + spi3 = 26, + i2s0 = 27, + i2s1 = 28, + i2s2 = 29, + uhci0 = 30, + uart0 = 31, + uart1 = 32, + uart2 = 33, + uart3 = 34, + uart4 = 35, + lcd_cam = 36, + adc = 37, + pwm0 = 38, + pwm1 = 39, + twai0 = 40, + twai1 = 41, + twai2 = 42, + rmt = 43, + i2c0 = 44, + i2c1 = 45, + tg0_t0 = 46, + tg0_t1 = 47, + tg0_wdt_level = 48, + tg1_t0 = 49, + tg1_t1 = 50, + tg1_wdt_level = 51, + ledc = 52, + systimer_target0 = 53, + systimer_target1 = 54, + systimer_target2 = 55, + ahb_pdma_in_ch0 = 56, + ahb_pdma_in_ch1 = 57, + ahb_pdma_in_ch2 = 58, + ahb_pdma_out_ch0 = 59, + ahb_pdma_out_ch1 = 60, + ahb_pdma_out_ch2 = 61, + axi_pdma_in_ch0 = 62, + axi_pdma_in_ch1 = 63, + axi_pdma_in_ch2 = 64, + axi_pdma_out_ch0 = 65, + axi_pdma_out_ch1 = 66, + axi_pdma_out_ch2 = 67, + rsa = 68, + aes = 69, + sha = 70, + ecc = 71, + ecdsa = 72, + km = 73, + gpio_intr0 = 74, + gpio_intr1 = 75, + gpio_intr2 = 76, + gpio_intr3 = 77, + gpio_pad_comp = 78, + from_cpu_intr0 = 79, + from_cpu_intr1 = 80, + from_cpu_intr2 = 81, + from_cpu_intr3 = 82, + cache = 83, + mspi = 84, + csi_bridge = 85, + dsi_bridge = 86, + csi = 87, + dsi = 88, + gmii_phy = 89, + lpi = 90, + pmt = 91, + eth_mac = 92, + usb_otg = 93, + usb_otg_endp_multi_proc = 94, + jpeg = 95, + ppa = 96, + core0_trace = 97, + core1_trace = 98, + hp_core_ctrl = 99, + isp = 100, + i3c_mst = 101, + i3c_slv = 102, + usb_otg11_ch0 = 103, + dma2d_in_ch0 = 104, + dma2d_in_ch1 = 105, + dma2d_out_ch0 = 106, + dma2d_out_ch1 = 107, + dma2d_out_ch2 = 108, + psram_mspi = 109, + hp_sysreg = 110, + pcnt = 111, + hp_pau = 112, + hp_parlio_rx = 113, + hp_parlio_tx = 114, + h264_dma2d_out_ch0 = 115, + h264_dma2d_out_ch1 = 116, + h264_dma2d_out_ch2 = 117, + h264_dma2d_out_ch3 = 118, + h264_dma2d_out_ch4 = 119, + h264_dma2d_in_ch0 = 120, + h264_dma2d_in_ch1 = 121, + h264_dma2d_in_ch2 = 122, + h264_dma2d_in_ch3 = 123, + h264_dma2d_in_ch4 = 124, + h264_dma2d_in_ch5 = 125, + h264_reg = 126, + assist_debug = 127, + + /// Chip rev >= 3.0 only - absent on this die. See the note above. + dma2d_in_ch2 = 133, + /// Chip rev >= 3.0 only - absent on this die. + dma2d_out_ch3 = 134, + /// Chip rev >= 3.0 only - absent on this die. + axi_perf_mon = 135, + + /// True on a source that this pre-v3 silicon does not have. + pub inline fn isRev3Only(self: Source) bool { + return @intFromEnum(self) >= 133; + } +}; + +/// The last source ID with a mapping register on pre-v3 silicon. +pub const max_source_id: u8 = @intFromEnum(Source.assist_debug); + +/// Core 0's interrupt matrix. Core 1's is 0x800 above it (`reg_base.h:198-199`) and is not reachable +/// from here: this image runs core 0 only - core 1 is held in reset at power-on +/// (HP_SYS_CLKRST REG_RST_EN_CORE1_GLOBAL defaults to 1) - and routing a source to a core that is +/// not running is a way to lose an interrupt silently rather than loudly. +const matrix_base: u32 = mmio.addr(regs.DR_REG_INTERRUPT_CORE0_BASE); + +/// The mapping register's only field: 6 bits, holding a CLIC ID. Taken from UART0's macro pair +/// because the field is identical in all 128 of them - `interrupt_core0_reg.h` repeats +/// `_INT_MAP` / mask 0x3F / shift 0 for every source. (`INTERRUPT_CORE0_*_INT_MAP_M` is one of the +/// 153 `_M` macros that are broken C inside ESP-IDF and appear here as poisoned decls; the `_S`/`_V` +/// pair is the only usable form, which is what `mmio.Field.of` takes.) +const int_map = Field.of(regs.INTERRUPT_CORE0_UART0_INT_MAP_S, regs.INTERRUPT_CORE0_UART0_INT_MAP_V); + +inline fn mapReg(source_id: u8) Reg { + return Reg.atAddress(matrix_base + 4 * @as(u32, source_id)); +} + +/// Point a peripheral source at an external CLIC line. +/// +/// This is only the matrix half. A routed source still needs `setEnabled(line, true)`, a trigger +/// type, a priority above the threshold, a handler, and mstatus.MIE - `configureLine` does the +/// CLIC-side four in the order the hardware wants. +/// +/// Several sources may share one line; that is the normal way to fit 128 sources into 32 lines, and +/// the handler then has to ask each peripheral whether it was the one. Nothing here prevents it. +pub fn route(source: Source, line: u5) void { + routeId(@intFromEnum(source), line); +} + +/// `route` by raw source ID, for a source this enum does not name. +/// +/// The write is a read-modify-write of the low 6 bits, exactly as `interrupt_clic_ll.h:46` does it +/// (`REG_SET_BITS(DR_REG_INTERRUPT_CORE0_BASE + 4*intr_src, intr_num, RV_INT_MASK)` with +/// RV_INT_MASK 63 at line 25). The upper 26 bits are reserved and preserved. +pub fn routeId(source_id: u8, line: u5) void { + std.debug.assert(source_id <= max_source_id); + mapReg(source_id).modify(.{int_map.is(@as(u32, line) + ext_offset)}); +} + +/// Detach a source from every line. +/// +/// Writes CLIC ID 0, which is `ETS_INVALID_INUM` on this chip (`soc/esp32p4/include/soc/soc.h:251`) +/// and is what `esp_system/port/cpu_start.c:185` writes into all 128 mapping registers at boot. +/// ID 0 is an internal RISC-V interrupt line that the matrix cannot actually drive, so it means +/// "nowhere" rather than "line 0" - note the asymmetry with `route`, which adds 16. +pub fn unroute(source: Source) void { + mapReg(@intFromEnum(source)).modify(.{int_map.is(0)}); +} + +/// Which external line a source is routed to, or null if it is unrouted or points at an internal ID. +pub fn routedLine(source: Source) ?u5 { + const id = mapReg(@intFromEnum(source)).get(int_map); + if (id < ext_offset or id >= ext_offset + line_count) return null; + return @intCast(id - ext_offset); +} + +// ------------------------------------------------------------------------- per-line control + +/// One 32-bit control word per CLIC ID at `DR_REG_CLIC_CTRL_BASE + 4*id` (`soc/clic_reg.h:69`). +/// Indexed by CLIC ID, so every accessor here adds `ext_offset` to the caller's line number. +/// +/// The same word is also described byte-wise by the `BYTE_CLIC_*` macros (clic_reg.h:113-160), and +/// ESP-IDF uses both spellings: `interrupt_clic_ll.h` does 32-bit REG_SET_FIELD, the TEE build does +/// 8-bit stores. They land on the same bits, and each field sits wholly inside one byte, so a +/// 32-bit read-modify-write of one field and a byte store of that byte are indistinguishable in the +/// resulting word. This file uses the 32-bit form throughout. +const clic_ctrl_base: u32 = mmio.addr(regs.DR_REG_CLIC_CTRL_BASE); + +/// Priority, bits [31:24]. Reset value 0x1f (clic_reg.h:70). +const int_ctl = Field.of(regs.CLIC_INT_CTL_S, regs.CLIC_INT_CTL_V); +/// Trigger type, bits [18:17]. +const int_attr_trig = Field.of(regs.CLIC_INT_ATTR_TRIG_S, regs.CLIC_INT_ATTR_TRIG_V); +/// Hardware vectoring: 1 means fetch the handler address from MTVT rather than trapping to mtvec. +const int_attr_shv = Field.of(regs.CLIC_INT_ATTR_SHV_S, regs.CLIC_INT_ATTR_SHV_V); +/// Enable, bit 8. +const int_ie = Field.of(regs.CLIC_INT_IE_S, regs.CLIC_INT_IE_V); +/// Pending, bit 0. Read/write, with asymmetric semantics - see `edgeAck`. +const int_ip = Field.of(regs.CLIC_INT_IP_S, regs.CLIC_INT_IP_V); + +inline fn ctrl(line: u5) Reg { + return Reg.atAddress(clic_ctrl_base + 4 * (@as(u32, line) + ext_offset)); +} + +/// By raw CLIC ID rather than by external line, for the one caller that has to reach the 16 +/// internal IDs: `init`, silencing everything the ROM may have left enabled. +inline fn ctrlRegById(clic_id: u32) Reg { + std.debug.assert(clic_id < total_ids); + return Reg.atAddress(clic_ctrl_base + 4 * clic_id); +} + +/// How a source drives its line. The encoding is a two-bit field whose *low* bit selects +/// level-versus-edge and whose high bit selects the edge, which is why `interrupt_clic_ll.h:60` +/// masks the read with `& 1` to answer "is it edge-triggered": `0b10` is a level interrupt too. +/// (`soc/clic_reg.h:84-88`.) +pub const Trigger = enum(u2) { + level = 0, + rising_edge = 1, + /// 0b10 - low bit clear, so this is a *level* trigger despite the encoding's shape. Present + /// only because the field is two bits wide; no source should be configured with it. + level_alias = 2, + falling_edge = 3, + + pub inline fn isEdge(self: Trigger) bool { + return @intFromEnum(self) & 1 != 0; + } +}; + +pub fn setEnabled(line: u5, on: bool) void { + ctrl(line).modify(.{int_ie.is(@intFromBool(on))}); +} + +pub fn isEnabled(line: u5) bool { + return ctrl(line).get(int_ie) == 1; +} + +pub fn setTrigger(line: u5, t: Trigger) void { + ctrl(line).modify(.{int_attr_trig.is(@intFromEnum(t))}); +} + +pub fn getTrigger(line: u5) Trigger { + return @enumFromInt(ctrl(line).get(int_attr_trig)); +} + +/// Priority 0..7, stored left-aligned in the 8-bit CLIC_INT_CTL field. +/// +/// The stored byte is `priority << (8 - NLBITS)` with the low bits **zero**, which is what +/// `esp_tee_rv_utils.h:112` writes and what `interrupt_clic_ll.h:74` reads back with `>> (8-NLBITS)`. +/// Note the asymmetry with the *threshold*, where IDF fills the same low bits with ones +/// (`csr_clic.h:59`). Copying the threshold's encoding here would leave a different word behind +/// than IDF's, for the same nominal priority. +pub fn setPriority(line: u5, priority: u3) void { + ctrl(line).modify(.{int_ctl.is(@as(u32, priority) << nlbits_shift)}); +} + +pub fn getPriority(line: u5) u3 { + return @intCast(ctrl(line).get(int_ctl) >> nlbits_shift); +} + +/// Hardware vectoring for one line. With SHV set, the CLIC jumps to `MTVT + 4*id` instead of to +/// mtvec's base; `installVectorTable` fills every slot with the same trap entry, so flipping this +/// changes the fetch path and not the code that runs. `interrupt_clic_ll.h:99-102`. +pub fn setVectored(line: u5, on: bool) void { + ctrl(line).modify(.{int_attr_shv.is(@intFromBool(on))}); +} + +pub fn isVectored(line: u5) bool { + return ctrl(line).get(int_attr_shv) == 1; +} + +pub fn isPending(line: u5) bool { + return ctrl(line).get(int_ip) == 1; +} + +/// Acknowledge an edge-triggered interrupt. +/// +/// Writing **1** to IP is what clears it for an edge source. That reads backwards, and clic_reg.h +/// only hints at it - "This bit has different set and clear logic in the case of level interrupt +/// and edge interrupt" (clic_reg.h:106-107) - but ESP-IDF's function that does exactly this store is +/// named `rv_utils_intr_edge_ack` (`esp_private/interrupt_clic.h`, the `REG_SET_BIT(..., CLIC_INT_IP)` +/// at the end of that header). For a *level* source this instead asserts the pending bit, which is +/// how software raises one by hand; there is no acknowledge for a level source at the CLIC at all, +/// the handler must clear the peripheral's own status register. +pub fn edgeAck(line: u5) void { + ctrl(line).modify(.{int_ip.is(1)}); +} + +/// Raise a line from software. Same store as `edgeAck`; the two names exist because the hardware +/// gives one write two meanings depending on `Trigger`. +pub fn setPending(line: u5) void { + ctrl(line).modify(.{int_ip.is(1)}); +} + +/// Bitmask of the 32 external lines that are enabled, one loop over the control words. Mirrors +/// `rv_utils_intr_get_enabled_mask` in `esp_private/interrupt_clic.h`. +pub fn enabledMask() u32 { + var m: u32 = 0; + var i: u5 = 0; + while (true) : (i += 1) { + if (isEnabled(i)) m |= @as(u32, 1) << i; + if (i == line_count - 1) break; + } + return m; +} + +// ----------------------------------------------------------------------------- the threshold + +/// CLIC_INT_THRESH_REG - 0x2080_0008 (`soc/clic_reg.h:61`), **not** the `mintthresh` CSR. See the +/// module comment: on this pre-v3 die `csr_clic.h` does not even define MINTTHRESH_CSR, and a write +/// to CSR 0x347 here is accepted and ignored. +const thresh_reg = Reg.at(regs.CLIC_INT_THRESH_REG); +const cpu_int_thresh = Field.of(regs.CLIC_CPU_INT_THRESH_S, regs.CLIC_CPU_INT_THRESH_V); + +/// Mask every interrupt whose priority is <= `level`. +/// +/// The comparison is **inclusive**: threshold 0 lets priorities 1..7 through, threshold 7 masks +/// everything. `esp_private/interrupt_clic.h:198-203` makes the same point when it computes +/// `mask_int_level_lower_than(n)` as `set_intlevel(n - 1)`. Reset is 0, i.e. open. +/// +/// Two details reproduced from IDF rather than invented: +/// * the byte is `(level << 5) | 0x1f` - the low `8 - NLBITS` bits are filled with **ones** +/// (`csr_clic.h:59`, NLBITS_TO_BYTE), which is the opposite of the per-line priority encoding; +/// * the register is read back immediately afterwards. That is not a paranoid verification, it is +/// ordering: `esp_private/interrupt_clic.h:139-144` records that the CPU does not see the new +/// threshold until the store has actually left the write buffer, and that a load - or about +/// eight nops - is what forces it. Without the load, re-enabling mstatus.MIE on the next +/// instruction can take an interrupt the new threshold was meant to mask. +/// +/// `write` rather than `modify` is deliberate and matches IDF's `REG_WRITE`: CLIC_CPU_INT_THRESH is +/// the register's only field, so there is nothing to preserve. +pub fn setThreshold(level: u3) void { + thresh_reg.write(.{cpu_int_thresh.is((@as(u32, level) << nlbits_shift) | nlbits_pad)}); + _ = thresh_reg.raw(); +} + +pub fn getThreshold() u3 { + return @intCast(thresh_reg.get(cpu_int_thresh) >> nlbits_shift); +} + +// ------------------------------------------------------------- vector table and trap entry + +/// CSR numbers, from `components/riscv/include/riscv/csr_clic.h`: +/// * `MTVT_CSR 0x307` (line 34) - base of the interrupt jump table. +/// * `MTVEC_MODE_CSR 3` (line 22) - the two low bits of mtvec that put the core in CLIC mode. +/// * `MINTSTATUS_CSR 0x346` (`soc/interrupt_reg.h:36`) - **non-standard on this die**; the RISC-V +/// CLIC specification and IDF's rev-3 path both say 0xFB1 (`csr_clic.h:40`). +/// * `MINTTHRESH_CSR 0x347` exists only when INTTHRESH_STANDARD is 1, which it is not here. +pub const mtvt_csr = 0x307; +pub const mintstatus_csr = 0x346; +pub const mtvec_mode_clic = 3; +/// mstatus.MIE. Same bit `clkrst.Guard` manipulates. +const mstatus_mie: u32 = 1 << 3; + +/// A line's handler. Runs with mstatus.MIE clear - this file does not implement nesting - on the +/// interrupted stack, so it must be short and must not use floating point: `trapEntry` saves the +/// integer caller-saved registers and nothing else, and `_start` leaves the FPU enabled, so a +/// handler that touches an f-register corrupts whatever it interrupted. +pub const Handler = *const fn (line: u5) void; + +var handlers: [line_count]?Handler = @splat(null); + +/// Interrupts that arrived on a line with no handler, or on one of the 16 internal CLIC IDs. Not +/// reset by anything here: a non-zero value after a run is the diagnostic. +pub var spurious: u32 = 0; + +/// The CLIC's jump table: one address per CLIC ID, internal and external. +/// +/// 48 entries, and 256-byte aligned because the CLIC requires MTVT to be aligned to a power of two +/// at least as large as the table (4 * 48 = 192 bytes, so 256). The alignment travels with the +/// symbol, so the generated linker script's `.bss ... ALIGN(4)` is not a problem - the linker pads +/// to the input section's own alignment. No dedicated section is needed and build.zig is unchanged. +/// +/// Every slot points at the same `trapEntry`. A per-line stub would save the dispatch load, but it +/// would be 48 near-identical pieces of assembly to be wrong in, and the win is a handful of cycles +/// against a handler call. The table exists because the hardware needs one when SHV is set, not +/// because the entries differ. +var vector_table: [total_ids]u32 align(256) = @splat(0); + +/// What `init` found before it changed anything. Diagnostics, and the only record of the state the +/// bootloader hands over in - every one of these is overwritten by `init` itself, so nothing else +/// can observe them. +pub var boot_state: BootState = .{}; +pub const BootState = struct { + /// mstatus.MIE as handed over. Measured 1 on this board, which is the fact the whole ownership + /// sequence below exists for. + mie: bool = false, + /// Which of the 32 external lines had CLIC_INT_IE set before `init` cleared them. + enabled_lines: u32 = 0, + /// How many of the 128 peripheral sources were pointing at an external line before `init` + /// detached them. + routed_sources: u32 = 0, +}; + +/// Take ownership of the interrupt controller, then point it at this file. +/// +/// **The bootloader hands over with interrupts globally enabled.** Measured: `mie_at_boot=1`. That +/// single fact is why this function is a sequence rather than three CSR writes, and it cost two +/// silent hangs to establish. Two separate hazards follow from it, and clearing MIE only fixes the +/// first: +/// +/// 1. `init(); attach(...)` used to take an interrupt the moment the line's IE bit went up, before +/// the caller had said it was ready. `globalDisable()` first fixes that. +/// +/// 2. **Whatever the ROM had armed is still armed.** The ROM ran with its own mtvec and its own +/// reasons to enable interrupts; the matrix and the CLIC's IE bits are not reset by the handover. +/// The instant this file's caller sets MIE, any line the ROM left enabled vectors into +/// `trapEntry` - on an ID nothing here has a handler for. That increments `spurious` and +/// `mret`s; and if the source is level-triggered and still asserting, the next instruction traps +/// again, forever, with the console silent. The failure looks exactly like "our own line is not +/// being delivered", which is what it was mistaken for. +/// +/// So this function does what ESP-IDF's `core_intr_matrix_clear` does before it trusts the +/// controller (`esp_system/port/cpu_start.c:174-198`), and in the same order: +/// * detach all 128 sources by writing ETS_INVALID_INUM (cpu_start.c:183-189); +/// * clear every line's enable, which IDF gets for free from the CLIC's reset values and this +/// image does not, because the ROM ran first; +/// * set every external line vectored (cpu_start.c:193-196 - "Set all the CPU interrupt lines to +/// vectored by default, as it is on other RISC-V targets"). +/// +/// The register differential could not have found any of this: MIE is a CSR, and the boot state of +/// the matrix is identical on both sides of every comparison because both sides inherit it. +/// +/// Leaves MIE clear. Enabling interrupts stays the caller's decision, via `globalEnable()`. +pub fn init() void { + boot_state.mie = globalEnabled(); + globalDisable(); + + // Record and then silence every line, before anything can be delivered anywhere. + var l: u5 = 0; + while (true) : (l += 1) { + if (isEnabled(l)) boot_state.enabled_lines |= @as(u32, 1) << l; + if (l == line_count - 1) break; + } + // All 48 IDs, internal ones included: this core's interrupts are ours now, and an internal ID + // left enabled is as capable of trapping into `trapEntry` as an external one. + var id: u32 = 0; + while (id < total_ids) : (id += 1) { + ctrlRegById(id).modify(.{int_ie.is(0)}); + } + + // Detach every source. cpu_start.c:183-189 writes ETS_INVALID_INUM (0) to all of them. + var src: u32 = 0; + while (src <= max_source_id) : (src += 1) { + const r = mapReg(@intCast(src)); + const was = r.get(int_map); + if (was >= ext_offset and was < ext_offset + line_count) boot_state.routed_sources += 1; + r.modify(.{int_map.is(0)}); + } + + const entry = @intFromPtr(&trapEntry); + for (&vector_table) |*slot| slot.* = @intCast(entry); + + asm volatile ("csrw %[csr], %[val]" + : + : [csr] "i" (mtvt_csr), + [val] "r" (@as(u32, @intCast(@intFromPtr(&vector_table)))), + ); + // mtvec = base | 3. Mode 3 is what `rv_utils_set_mtvec` writes (`riscv/rv_utils.h:168-171` with + // MTVEC_MODE_CSR from `csr_clic.h:22`) and it is what makes the core interpret mcause and MTVT + // as CLIC rather than as the standard vectored interface. + // + // The hardware uses `mtvec[31:6] << 6` (vectors_clic.S:38-46 spells this out), so it ignores the + // low six bits entirely: a `trapEntry` that were not 64-byte aligned would silently vector up to + // 60 bytes *before* the function. `trapEntryAddress()` exists so a test can prove on the die + // that it is aligned rather than trusting the linker. + asm volatile ("csrw mtvec, %[val]" + : + : [val] "r" (@as(u32, @intCast(entry)) | mtvec_mode_clic), + ); + + // Every external line vectored, matching cpu_start.c:193-196. Also the safer default in its own + // right: SHV=1 is the only delivery path ESP-IDF exercises on this chip, so it is the only one + // the silicon has been validated against. See `configureLine`. + l = 0; + while (true) : (l += 1) { + setVectored(l, true); + if (l == line_count - 1) break; + } + + // Threshold open, matching IDF's RVHAL_INTR_ENABLE_THRESH of 0 (`csr_clic.h:16`): every line + // then gates on its own IE bit and its priority, which is where a driver can reason about it. + setThreshold(0); +} + +/// Diagnostics a behavioural test can print, because the two facts they establish - that the trap +/// entry is 64-byte aligned and that MTVT is 256-byte aligned - are properties of the *link*, and +/// the shipped image is stripped, so there is no way to check them from the host. +pub fn trapEntryAddress() u32 { + return @intCast(@intFromPtr(&trapEntry)); +} + +pub fn vectorTableAddress() u32 { + return @intCast(@intFromPtr(&vector_table)); +} + +pub fn readMtvec() u32 { + return asm volatile ("csrr %[out], mtvec" + : [out] "=r" (-> u32), + ); +} + +pub fn readMtvt() u32 { + return asm volatile ("csrr %[out], %[csr]" + : [out] "=r" (-> u32), + : [csr] "i" (mtvt_csr), + ); +} + +/// mintstatus, CSR 0x346 on this die (`soc/interrupt_reg.h:36`). Bits [31:24] are the current +/// interrupt level: non-zero outside a handler would mean a previous trap never returned. +pub fn readMintstatus() u32 { + return asm volatile ("csrr %[out], %[csr]" + : [out] "=r" (-> u32), + : [csr] "i" (mintstatus_csr), + ); +} + +/// Install (or, with null, remove) the handler for one external line. +/// +/// Done with interrupts masked because the store is a pointer the trap entry may be about to load; +/// `clkrst.maskInterrupts` composes - it restores only the MIE that was there - so this is safe to +/// call from inside an already-masked region. +pub fn setHandler(line: u5, handler: ?Handler) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + handlers[line] = handler; +} + +/// Everything one line needs, in the order the hardware wants: handler before enable, so a source +/// that is already pending cannot reach an empty slot; trigger and priority before enable, so the +/// first interrupt is taken under the intended configuration rather than under the reset one. +/// +/// Does not touch the matrix - `route` is the other half - and does not touch mstatus. +pub fn configureLine(line: u5, opts: struct { + handler: Handler, + trigger: Trigger = .level, + /// Must exceed the threshold to ever be taken; the threshold comparison is inclusive. + priority: u3 = 1, + /// Hardware vectoring: fetch the handler address from `MTVT + 4*id` instead of trapping to + /// mtvec's base. + /// + /// **On by default, and the default is the interesting part.** Every slot of the table holds the + /// same `trapEntry`, so this changes only how the core finds that address - which makes the + /// choice look free, and it is not. ESP-IDF sets SHV on all 32 lines at boot + /// (`cpu_start.c:193-196`, "Set all the CPU interrupt lines to vectored by default, as it is on + /// other RISC-V targets") and puts nothing but `j _panic_handler` at mtvec's base + /// (`vectors_clic.S:47-52`). So on this chip the SHV=0 delivery path is one ESP-IDF never takes + /// and therefore one nobody has validated. Defaulting to the path the vendor exercises is worth + /// more than the memory fetch it costs. + /// + /// **Measured on the die: it is the other way round, and the default is now `false`.** + /// + /// With SHV=1 the interrupt was never delivered. The core vectored to a wild address and took an + /// instruction access fault - `mcause=0x30000001` (EXCCODE 1, MINHV clear, so the fault was not + /// during the table fetch), at a `mepc` that differed run to run, with `taken=0` proving the + /// trap entry was never reached. mtvec, MTVT and the table contents were all verified correct + /// beforehand: `mtvec=0x40001383` = entry|3, `mtvt=0x4ff00100`, and every slot holding + /// `0x40001380` = `trapEntry`. + /// + /// The difference from ESP-IDF is *where the table lives*. IDF's `_mtvt_table` is in + /// `.section .exception_vectors_table.text` (`vectors_clic.S:32,67`), i.e. instruction space. + /// This image has no IRAM: it executes from flash through the MMU, so a table that `init()` has + /// to write must live in L2MEM, and the hardware vector fetch does not appear to work from + /// there. Since flash is not writable at run time, there is nowhere else to put it, which makes + /// SHV=0 the correct choice for this memory layout rather than a workaround. + /// + /// With SHV=0 both halves of the behavioural test pass: one interrupt taken, dispatched to the + /// right handler, `last_clic_id=21`, no spurious - and the threshold experiment then shows the + /// memory-mapped register at 0x2080_0008 really is the one the arbiter reads. + /// + /// `true` remains available for an image that gains an IRAM section, and the vector table is + /// still populated so that switching is a one-word change. + vectored: bool = false, +}) void { + setHandler(line, opts.handler); + setTrigger(line, opts.trigger); + setPriority(line, opts.priority); + setVectored(line, opts.vectored); + setEnabled(line, true); +} + +/// Route a source and bring its line up in one call. +pub fn attach(source: Source, line: u5, opts: struct { + handler: Handler, + trigger: Trigger = .level, + priority: u3 = 1, + /// See `configureLine`: vectored is the only path ESP-IDF exercises on this chip. + vectored: bool = false, +}) void { + route(source, line); + configureLine(line, .{ + .handler = opts.handler, + .trigger = opts.trigger, + .priority = opts.priority, + .vectored = opts.vectored, + }); +} + +// --------------------------------------------------------------------------- global enable + +/// mstatus.MIE on. Nothing is taken before this, whatever the CLIC is configured to do. +pub inline fn globalEnable() void { + asm volatile ("csrs mstatus, %[m]" + : + : [m] "r" (mstatus_mie), + ); +} + +pub inline fn globalDisable() void { + asm volatile ("csrc mstatus, %[m]" + : + : [m] "r" (mstatus_mie), + ); +} + +pub inline fn globalEnabled() bool { + const s = asm volatile ("csrr %[out], mstatus" + : [out] "=r" (-> u32), + ); + return s & mstatus_mie != 0; +} + +/// The composable form: mask, do something, restore whatever was there. +/// +/// const guard = intr.mask(); +/// defer guard.release(); +/// +/// This is `clkrst.maskInterrupts` under another name, re-exported rather than reimplemented so +/// that a critical section written against either module is the same critical section. It nests +/// correctly - `release` only sets MIE if MIE was set on entry - which is why `setHandler` can use +/// it without caring who called it. +pub const Guard = clkrst.Guard; +pub inline fn mask() Guard { + return clkrst.maskInterrupts(); +} + +// ------------------------------------------------------------------------------- trap entry + +/// How many times `trapEntry` has dispatched an interrupt, and the last CLIC ID it saw. Diagnostics: +/// with `taken == 0` the trap was never reached at all, which separates "the CLIC did not deliver" +/// from "the handler did not run". +pub var taken: u32 = 0; +pub var last_clic_id: u32 = 0; + +/// An exception - not an interrupt - that reached `trapEntry`. +pub const Fault = struct { + /// Full mcause. Bit 31 is clear by construction here; the low bits are the exception code + /// (1 instruction access, 2 illegal instruction, 5 load access, 7 store access, 11 ecall). + mcause: u32, + /// The instruction that faulted. + mepc: u32, + /// The address or instruction word involved, per exception code. + mtval: u32, +}; + +pub var faults: u32 = 0; +pub var last_fault: Fault = .{ .mcause = 0, .mepc = 0, .mtval = 0 }; + +/// Called with the fault already recorded, before parking. Install one to get the numbers out; +/// `hal` cannot print, so this hook is the only way a fault becomes visible. +/// +/// hal.intr.on_fault = struct { +/// fn f(x: hal.intr.Fault) void { +/// soc.rom.print("MARK FAULT mcause=0x%08x mepc=0x%08x mtval=0x%08x\r\n", +/// .{ x.mcause, x.mepc, x.mtval }); +/// } +/// }.f; +pub var on_fault: ?*const fn (Fault) void = null; + +/// Called from `trapEntry` with the CLIC ID out of mcause. Not part of the API; `export` because +/// the assembly calls it by name. +export fn intrDispatch(clic_id: u32) callconv(.c) void { + taken +%= 1; + last_clic_id = clic_id; + if (clic_id < ext_offset or clic_id >= ext_offset + line_count) { + // One of the 16 internal IDs. This file routes nothing there, so it is a bug elsewhere - + // most likely something the ROM left armed that `init` did not manage to silence. + spurious +%= 1; + return; + } + const line: u5 = @intCast(clic_id - ext_offset); + if (handlers[line]) |h| h(line) else spurious +%= 1; +} + +/// The exception arm of `trapEntry`. Records, reports if a hook is installed, and **parks**. +/// +/// Parking rather than returning is the whole point. `mret` from an exception resumes at the +/// faulting instruction, which faults again immediately: every mistake anywhere in this file used to +/// become an unbreakable loop through the trap entry with the console silent, indistinguishable from +/// "the interrupt was never delivered". It cost a debugging round to tell those apart. ESP-IDF makes +/// the same choice by putting `j _panic_handler` at mtvec's base (`vectors_clic.S:47-52`). +export fn intrFault(mcause: u32, mepc: u32, mtval: u32) callconv(.c) noreturn { + faults +%= 1; + last_fault = .{ .mcause = mcause, .mepc = mepc, .mtval = mtval }; + globalDisable(); + if (on_fault) |f| f(last_fault); + while (true) {} +} + +/// The trap entry: every trap on this core arrives here, interrupt or exception. +/// +/// Reached three ways, and they are not interchangeable: +/// * an **interrupt with SHV = 1**, through `MTVT + 4*id`; +/// * an **interrupt with SHV = 0**, through mtvec's base; +/// * an **exception**, always through mtvec's base, whatever any line's SHV says. +/// +/// 64-byte aligned, and this is a hardware requirement rather than tidiness: in CLIC mode the core +/// computes the target as `mtvec[31:6] << 6` (`vectors_clic.S:38-46` states it outright), so the low +/// six bits of mtvec are not part of the address. A trap entry that were not 64-byte aligned would +/// vector up to 60 bytes *before* this function, into whatever the linker put there. Measured in the +/// linked image: 0x4000_1140, and `trapEntryAddress()` lets a test confirm it on the die, since the +/// shipped image is stripped and there is no symbol to check from the host. +/// +/// **The first thing it does is decide whether this was an interrupt at all.** mcause bit 31 says +/// so, and getting that wrong is not a small bug: an exception whose handler `mret`s resumes at the +/// faulting instruction and faults again, immediately and forever, with the console silent. That +/// failure is indistinguishable from "the interrupt was never delivered", and the two were in fact +/// confused for a debugging round. So the exception arm never returns - see `intrFault`. +/// +/// Saves the integer caller-saved set - ra, t0-t6, a0-a7, sixteen words - and nothing else. Not +/// saved, deliberately and with consequences: +/// * **the f registers.** `src/main.zig`'s `_start` sets mstatus.FS to enable the FPU, so a handler +/// that does float arithmetic silently corrupts the interrupted code. Handlers must stay integer. +/// * **mepc, mcause, mstatus.** In CLIC mode the core stacks the previous privilege, interrupt +/// enable and interrupt level in mcause itself, and `mret` restores them from there - so nothing +/// here may write mcause, and nothing does. They are only at risk from a *nested* trap, and MIE +/// stays clear for the whole sequence, so nothing can nest. That is also why there is no `mnxti` +/// loop: the CLIC's hardware nesting (SOC_INT_HW_NESTED_SUPPORTED, `soc_caps.h:193`) is unused. +/// +/// One consequence of `mret` worth stating because it defeats an obvious defence: it restores +/// mstatus.MIE from MPIE, which the hardware set to 1 on entry. A handler that calls +/// `globalDisable()` therefore does **not** leave interrupts off after it returns. To stop a runaway +/// source the handler must clear it at the peripheral, or call `setEnabled(line, false)`. +export fn trapEntry() align(64) callconv(.naked) noreturn { + asm volatile ( + \\ addi sp, sp, -64 + \\ sw ra, 0(sp) + \\ sw t0, 4(sp) + \\ sw t1, 8(sp) + \\ sw t2, 12(sp) + \\ sw a0, 16(sp) + \\ sw a1, 20(sp) + \\ sw a2, 24(sp) + \\ sw a3, 28(sp) + \\ sw a4, 32(sp) + \\ sw a5, 36(sp) + \\ sw a6, 40(sp) + \\ sw a7, 44(sp) + \\ sw t3, 48(sp) + \\ sw t4, 52(sp) + \\ sw t5, 56(sp) + \\ sw t6, 60(sp) + \\ csrr a0, mcause + // Bit 31 set means interrupt, so mcause read as *signed* is negative. `bgez` therefore + // branches exactly on "this was an exception", in one instruction and with no scratch + // register - which matters here because every scratch register is already spoken for. + \\ bgez a0, 1f + // mcause[11:0] is the CLIC's interrupt ID. Isolated with a shift pair rather than `andi`: + // andi's immediate is 12-bit *signed*, so `andi a0, a0, 0xfff` does not assemble as a + // 12-bit mask - it is -1, and would leave the interrupt bit and the level field in place. + \\ slli a0, a0, 20 + \\ srli a0, a0, 20 + \\ call intrDispatch + \\ lw ra, 0(sp) + \\ lw t0, 4(sp) + \\ lw t1, 8(sp) + \\ lw t2, 12(sp) + \\ lw a0, 16(sp) + \\ lw a1, 20(sp) + \\ lw a2, 24(sp) + \\ lw a3, 28(sp) + \\ lw a4, 32(sp) + \\ lw a5, 36(sp) + \\ lw a6, 40(sp) + \\ lw a7, 44(sp) + \\ lw t3, 48(sp) + \\ lw t4, 52(sp) + \\ lw t5, 56(sp) + \\ lw t6, 60(sp) + \\ addi sp, sp, 64 + \\ mret + // The exception arm. No restore and no `mret`: `intrFault` is noreturn, because resuming + // would re-execute the faulting instruction. The saved registers stay on the stack, which + // costs 64 bytes that are never reclaimed and is the correct trade for a path that ends in + // a parked core with the numbers printed. + \\1: + \\ csrr a1, mepc + \\ csrr a2, mtval + \\ call intrFault + ); +} + +// ------------------------------------------------------------------------------------ tests + +test "the enum's IDs are the offsets of the matrix registers they name" { + // The whole of `routeId` rests on `map_reg_addr == base + 4*id`. These four are checked against + // the addresses ESP-IDF's own interrupt_core0_reg.h computes, which is an independent path: + // IDF wrote the offset as a literal per source, this file multiplies. + try std.testing.expectEqual(@as(u32, 0x7c), 4 * @as(u32, @intFromEnum(Source.uart0))); + try std.testing.expectEqual(@as(u32, 0xb0), 4 * @as(u32, @intFromEnum(Source.i2c0))); + try std.testing.expectEqual(@as(u32, 0xd0), 4 * @as(u32, @intFromEnum(Source.ledc))); + try std.testing.expectEqual(@as(u32, 0x1fc), 4 * @as(u32, @intFromEnum(Source.assist_debug))); +} + +test "rev-3-only sources are flagged and the pre-v3 ones are not" { + try std.testing.expect(Source.axi_perf_mon.isRev3Only()); + try std.testing.expect(Source.dma2d_in_ch2.isRev3Only()); + try std.testing.expect(!Source.assist_debug.isRev3Only()); + try std.testing.expect(!Source.dma2d_in_ch1.isRev3Only()); +} + +test "priority and threshold use different encodings of the same three bits" { + // Priority pads low with zeros, threshold pads low with ones. Getting these the same way round + // is the mistake this test exists to catch. + const priority_byte = @as(u32, 5) << nlbits_shift; + const threshold_byte = (@as(u32, 5) << nlbits_shift) | nlbits_pad; + try std.testing.expectEqual(@as(u32, 0xa0), priority_byte); + try std.testing.expectEqual(@as(u32, 0xbf), threshold_byte); + try std.testing.expectEqual(@as(u32, 5), priority_byte >> nlbits_shift); + try std.testing.expectEqual(@as(u32, 5), threshold_byte >> nlbits_shift); +} + +test "trigger's low bit, not its value, decides edge versus level" { + try std.testing.expect(Trigger.rising_edge.isEdge()); + try std.testing.expect(Trigger.falling_edge.isEdge()); + try std.testing.expect(!Trigger.level.isEdge()); + try std.testing.expect(!Trigger.level_alias.isEdge()); +} + +test "the vector table is aligned to a power of two above its own size" { + try std.testing.expectEqual(@as(usize, 256), @alignOf(@TypeOf(vector_table))); + try std.testing.expect(@sizeOf(@TypeOf(vector_table)) <= 256); +} + +test "mcause's sign bit is what separates an interrupt from an exception" { + // The trap entry branches on `bgez mcause`, which is only correct if bit 31 is the interrupt + // flag and the value is read signed. Spelled out here because the asm cannot say it. + const interrupt_mcause: u32 = 0x8000_0015; // CLIC ID 21 = external line 5 + const exception_mcause: u32 = 0x0000_0002; // illegal instruction + try std.testing.expect(@as(i32, @bitCast(interrupt_mcause)) < 0); + try std.testing.expect(@as(i32, @bitCast(exception_mcause)) >= 0); + // And the ID extraction the two shifts perform. + try std.testing.expectEqual(@as(u32, 21), (interrupt_mcause << 20) >> 20); +} + +test "mtvec's mode bits do not collide with a 64-byte-aligned base" { + // The hardware target is `mtvec[31:6] << 6`, so the mode goes in bits the base cannot use - + // but only if the base really is 64-byte aligned. This is the arithmetic `init` performs; + // whether the *linked* trapEntry satisfies it is a fact about the link, and + // `trapEntryAddress()` is how a test on the die checks that, the image being stripped. + const aligned_base: u32 = 0x4000_1200; + const mtvec = aligned_base | mtvec_mode_clic; + try std.testing.expectEqual(aligned_base, (mtvec >> 6) << 6); + // A base one instruction short of alignment vectors 60 bytes early, silently. + const bad_base: u32 = 0x4000_1204; + try std.testing.expect(((bad_base | mtvec_mode_clic) >> 6) << 6 != bad_base); +} diff --git a/src/hal/ledc.zig b/src/hal/ledc.zig new file mode 100644 index 0000000..62eacd8 --- /dev/null +++ b/src/hal/ledc.zig @@ -0,0 +1,587 @@ +//! LEDC: the LED PWM controller. Four timers, eight channels, and the first **shadow-register** +//! peripheral in this HAL. +//! +//! Three things make LEDC different from everything else here, and all three are load-bearing. +//! +//! **1. Configuration is staged, then committed.** `LEDC_PARA_UP_CHn` (channel) and +//! `LEDC_TIMERn_PARA_UP` (timer) are write-to-trigger bits: writing 1 copies the staged fields into +//! the shadow registers the counter and comparators actually use, and the hardware clears the bit +//! again by itself (`ledc_reg.h:42-47`, `:951-958`). Values written without a commit are visible in +//! the register file and have no effect on the output. So every mutator here stages, and every +//! commit is its own store - `commitChannel` / `commitTimer` - exactly as ESP-IDF's +//! `ledc_ll_ls_channel_update` (ledc_ll.h:435-438) and `ledc_ll_ls_timer_update` (ledc_ll.h:286-290) +//! do it. +//! +//! The commit store is a read-modify-write, and that is deliberate rather than sloppy: the commit +//! bit shares its word with the staged fields it commits. `LEDC_PARA_UP_CH0` is bit 4 of +//! `LEDC_CH0_CONF0_REG`, whose other fields are `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV` and `OVF_*`, so +//! a bare `writeRaw(1 << 4)` would erase the very configuration it was meant to commit. Compare +//! `systimer.zig`'s `op.write(.{update.is(1)})`, which is a single whole-word store because +//! `SYSTIMER_UNIT0_OP_REG` contains nothing else. The read-modify-write is safe here for the reason +//! `mmio.zig` gives: `PARA_UP` is `WT`, it reads back 0, so the read half of the read-modify-write +//! can never re-trigger an earlier commit. That is the difference between a self-clearing bit and a +//! write-1-to-clear bit, and it is why LEDC does not need the interrupt-status treatment. +//! +//! **2. The divider is fixed point, Q10.8.** `LEDC_CLK_DIV_TIMERn` is an 18-bit field at [22:5] +//! (`ledc_reg.h:921-928`) holding a divider with 8 fractional bits (`LEDC_LL_FRACTIONAL_BITS`, +//! ledc_ll.h:30): bits [17:8] are the integer part, bits [7:0] the fraction, so the value 0x4E2 +//! means 1250/256 = 4.8828. The output frequency is +//! +//! f_pwm = f_src * 256 / (div * 2^duty_res) +//! +//! and `divisor()` below is ESP-IDF's arithmetic for the inverse, transcribed operation for +//! operation from `esp_driver_ledc/src/ledc.c:468-497` - including the two places where it is +//! surprising. See its comment. +//! +//! **3. On the P4 the clock mux left the peripheral.** `LEDC_CONF_REG.LEDC_APB_CLK_SEL` still exists +//! in the register map and still documents an encoding (0: APB, 1: RC_FAST, 2: XTAL), and ESP-IDF's +//! P4 LL never touches it: the real mux is `HP_SYS_CLKRST.PERI_CLK_CTRL22.REG_LEDC_CLK_SRC_SEL`, +//! with a *different* encoding (0: XTAL, 1: RC_FAST, 2: PLL_DIV) - ledc_ll.h:223-242. Writing the +//! in-block register would silently do nothing, and reading it back to check would silently agree. +//! `ClockSource` below is the HP_SYS_CLKRST encoding. +//! +//! Gamma fade *ramps* are out of scope, but one gamma register is not optional: the P4 moved +//! `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` out of `LEDC_CHn_CONF1_REG` - which on this die +//! holds only `DUTY_START` - and into gamma RAM. A constant duty is therefore a degenerate one-step +//! fade, and `setDuty` writes that single entry, which is what ESP-IDF's `ledc_duty_config` does for +//! every plain duty change (ledc.c:263-280). + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); +const gpio = @import("gpio.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// Eight channels, four timers (`soc_caps.h:385-386`). +pub const channel_count = 8; +pub const timer_count = 4; + +/// The counter is 20 bits, so the duty resolution is at most 20 (`soc_caps.h:387`). The register +/// field is five bits wide and will happily accept 21-31; the hardware will not. +pub const max_duty_resolution = 20; + +/// Fractional bits in `LEDC_CLK_DIV_TIMERn` - `LEDC_LL_FRACTIONAL_BITS`, ledc_ll.h:30. +pub const fractional_bits = 8; + +/// The divider must be at least 1.0 and must fit the field: ESP-IDF's `LEDC_IS_DIV_INVALID` +/// (ledc.c:114) rejects anything `<= LEDC_LL_FRACTIONAL_MAX` or `> LEDC_TIMER_DIV_NUM_MAX`. +pub const divisor_min: u32 = 1 << fractional_bits; +pub const divisor_max: u32 = 0x3ffff; + +pub const Error = error{ + /// The requested frequency cannot be reached from this source at this resolution: the divider + /// would be below 1.0 (frequency too high) or wider than 18 bits (frequency too low). + DividerOutOfRange, + DutyResolutionOutOfRange, +}; + +// ------------------------------------------------------------------------------------- registers + +// Five registers per channel, stride 0x14; two per timer, stride 0x08. Both strides come from a +// second instance's macro rather than being assumed - see mmio.RegArray. +const ch_conf0 = mmio.RegArray(regs.LEDC_CH0_CONF0_REG, regs.LEDC_CH1_CONF0_REG, channel_count); +const ch_hpoint = mmio.RegArray(regs.LEDC_CH0_HPOINT_REG, regs.LEDC_CH1_HPOINT_REG, channel_count); +const ch_duty = mmio.RegArray(regs.LEDC_CH0_DUTY_REG, regs.LEDC_CH1_DUTY_REG, channel_count); +const ch_conf1 = mmio.RegArray(regs.LEDC_CH0_CONF1_REG, regs.LEDC_CH1_CONF1_REG, channel_count); +const ch_duty_r = mmio.RegArray(regs.LEDC_CH0_DUTY_R_REG, regs.LEDC_CH1_DUTY_R_REG, channel_count); +const ch_gamma_conf = mmio.RegArray(regs.LEDC_CH0_GAMMA_CONF_REG, regs.LEDC_CH1_GAMMA_CONF_REG, channel_count); +// Gamma RAM: 16 entries per channel, so the per-channel stride is 0x40 and entry 0 is the base. +const ch_gamma_range0 = mmio.RegArray(regs.LEDC_CH0_GAMMA_RANGE0_REG, regs.LEDC_CH1_GAMMA_RANGE0_REG, channel_count); +const tim_conf = mmio.RegArray(regs.LEDC_TIMER0_CONF_REG, regs.LEDC_TIMER1_CONF_REG, timer_count); +const tim_value = mmio.RegArray(regs.LEDC_TIMER0_VALUE_REG, regs.LEDC_TIMER1_VALUE_REG, timer_count); + +// Field geometry is taken from instance 0 and reused for every instance, which is only sound if the +// instances agree; the comptime block below checks the ends of both ranges against instance 0. That +// is not paranoia about the silicon, it is paranoia about the macro names: `LEDC_CLK_DIV_TIMER0` and +// `LEDC_TIMER0_DUTY_RES` put the instance number in different places, and picking up +// `LEDC_TIMER1_DUTY_RES_S` while meaning timer 0's shift is a one-character mistake. +const timer_sel = Field.of(regs.LEDC_TIMER_SEL_CH0_S, regs.LEDC_TIMER_SEL_CH0_V); +const sig_out_en = Field.of(regs.LEDC_SIG_OUT_EN_CH0_S, regs.LEDC_SIG_OUT_EN_CH0_V); +const idle_lv = Field.of(regs.LEDC_IDLE_LV_CH0_S, regs.LEDC_IDLE_LV_CH0_V); +const ch_para_up = Field.of(regs.LEDC_PARA_UP_CH0_S, regs.LEDC_PARA_UP_CH0_V); +const hpoint = Field.of(regs.LEDC_HPOINT_CH0_S, regs.LEDC_HPOINT_CH0_V); +const duty = Field.of(regs.LEDC_DUTY_CH0_S, regs.LEDC_DUTY_CH0_V); +const duty_r = Field.of(regs.LEDC_DUTY_CH0_R_S, regs.LEDC_DUTY_CH0_R_V); +const duty_start = Field.of(regs.LEDC_DUTY_START_CH0_S, regs.LEDC_DUTY_START_CH0_V); +const gamma_entry_num = Field.of(regs.LEDC_CH0_GAMMA_ENTRY_NUM_S, regs.LEDC_CH0_GAMMA_ENTRY_NUM_V); +const gamma_duty_inc = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_INC_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_INC_V); +const gamma_duty_cycle = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_CYCLE_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_CYCLE_V); +const gamma_scale = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_SCALE_S, regs.LEDC_CH0_GAMMA_RANGE0_SCALE_V); +const gamma_duty_num = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_NUM_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_NUM_V); + +const duty_res = Field.of(regs.LEDC_TIMER0_DUTY_RES_S, regs.LEDC_TIMER0_DUTY_RES_V); +const clk_div = Field.of(regs.LEDC_CLK_DIV_TIMER0_S, regs.LEDC_CLK_DIV_TIMER0_V); +const tim_pause = Field.of(regs.LEDC_TIMER0_PAUSE_S, regs.LEDC_TIMER0_PAUSE_V); +const tim_rst = Field.of(regs.LEDC_TIMER0_RST_S, regs.LEDC_TIMER0_RST_V); +const tim_para_up = Field.of(regs.LEDC_TIMER0_PARA_UP_S, regs.LEDC_TIMER0_PARA_UP_V); + +comptime { + const same = struct { + fn check(comptime what: []const u8, comptime a: Field, comptime b: Field) void { + if (a.shift != b.shift or a.width != b.width) @compileError( + "the per-instance " ++ what ++ " macros disagree on bit position or width; " ++ + "this file must index the field per instance instead of reusing instance 0's", + ); + } + }.check; + // Channels: 1 and 7, the two ends of the range beyond instance 0. + same("LEDC_TIMER_SEL_CHn", timer_sel, Field.of(regs.LEDC_TIMER_SEL_CH1_S, regs.LEDC_TIMER_SEL_CH1_V)); + same("LEDC_TIMER_SEL_CHn", timer_sel, Field.of(regs.LEDC_TIMER_SEL_CH7_S, regs.LEDC_TIMER_SEL_CH7_V)); + same("LEDC_SIG_OUT_EN_CHn", sig_out_en, Field.of(regs.LEDC_SIG_OUT_EN_CH7_S, regs.LEDC_SIG_OUT_EN_CH7_V)); + same("LEDC_IDLE_LV_CHn", idle_lv, Field.of(regs.LEDC_IDLE_LV_CH7_S, regs.LEDC_IDLE_LV_CH7_V)); + same("LEDC_PARA_UP_CHn", ch_para_up, Field.of(regs.LEDC_PARA_UP_CH7_S, regs.LEDC_PARA_UP_CH7_V)); + same("LEDC_HPOINT_CHn", hpoint, Field.of(regs.LEDC_HPOINT_CH7_S, regs.LEDC_HPOINT_CH7_V)); + same("LEDC_DUTY_CHn", duty, Field.of(regs.LEDC_DUTY_CH7_S, regs.LEDC_DUTY_CH7_V)); + same("LEDC_DUTY_START_CHn", duty_start, Field.of(regs.LEDC_DUTY_START_CH7_S, regs.LEDC_DUTY_START_CH7_V)); + same("LEDC_CHn_GAMMA_ENTRY_NUM", gamma_entry_num, Field.of(regs.LEDC_CH7_GAMMA_ENTRY_NUM_S, regs.LEDC_CH7_GAMMA_ENTRY_NUM_V)); + same("LEDC_CHn_GAMMA_RANGE0_SCALE", gamma_scale, Field.of(regs.LEDC_CH7_GAMMA_RANGE0_SCALE_S, regs.LEDC_CH7_GAMMA_RANGE0_SCALE_V)); + // Timers: 1 and 3. + same("LEDC_TIMERn_DUTY_RES", duty_res, Field.of(regs.LEDC_TIMER1_DUTY_RES_S, regs.LEDC_TIMER1_DUTY_RES_V)); + same("LEDC_TIMERn_DUTY_RES", duty_res, Field.of(regs.LEDC_TIMER3_DUTY_RES_S, regs.LEDC_TIMER3_DUTY_RES_V)); + same("LEDC_CLK_DIV_TIMERn", clk_div, Field.of(regs.LEDC_CLK_DIV_TIMER3_S, regs.LEDC_CLK_DIV_TIMER3_V)); + same("LEDC_TIMERn_PAUSE", tim_pause, Field.of(regs.LEDC_TIMER3_PAUSE_S, regs.LEDC_TIMER3_PAUSE_V)); + same("LEDC_TIMERn_RST", tim_rst, Field.of(regs.LEDC_TIMER3_RST_S, regs.LEDC_TIMER3_RST_V)); + same("LEDC_TIMERn_PARA_UP", tim_para_up, Field.of(regs.LEDC_TIMER3_PARA_UP_S, regs.LEDC_TIMER3_PARA_UP_V)); + + // `LEDC_TIMER_DIV_NUM_MAX` (ledc.c:110) is a literal in the driver; it should be the field's + // own mask, and if a future die widens the field this is where the two part company. + if (divisor_max != clk_div.max()) @compileError( + "divisor_max no longer matches LEDC_CLK_DIV_TIMERn's width", + ); + // The eight output signals must be consecutive for `signalIndex` to be arithmetic. + if (regs.LEDC_LS_SIG_OUT_PAD_OUT7_IDX - regs.LEDC_LS_SIG_OUT_PAD_OUT0_IDX != channel_count - 1) + @compileError("the LEDC output signal indices are not consecutive; signalIndex must be a table"); +} + +// ------------------------------------------------------------------------------ clocks and reset + +/// LEDC's function clock, in HP_SYS_CLKRST rather than in the peripheral (ledc_ll.h:179, :241). +/// Shared with RMT's fields, hence the interrupt-masked read-modify-write. +const peri_clk_ctrl22 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL22_REG); +const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_LEDC_CLK_SRC_SEL_V); +const func_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_CLK_EN_S, regs.HP_SYS_CLKRST_REG_LEDC_CLK_EN_V); + +/// The four timers' shared source. Encoding from `ledc_ll_set_slow_clk_sel` (ledc_ll.h:223-242) - +/// *not* the encoding `LEDC_CONF_REG.APB_CLK_SEL` documents, which is a different register on a +/// different block and is dead on this die. +pub const ClockSource = enum(u2) { + /// 40 MHz on this board (`clk_tree_defs.h:145`). + xtal = 0, + /// The internal RC oscillator: approximately 17.5 MHz (`clk_tree_defs.h:58`) and not trimmed. + /// ESP-IDF calibrates it against XTAL before using it for a divider; there is no calibration + /// here, so a frequency computed from `rc_fast_hz_approx` is approximate too. + rc_fast = 1, + /// PLL_F80M, 80 MHz (`clk_tree_defs.h:168`). Called `LEDC_SLOW_CLK_PLL_DIV` by ESP-IDF. + pll_div = 2, + + /// The source frequency to feed `divisor`, or null for RC_FAST, whose real rate has to be + /// measured rather than assumed. + pub fn hz(self: ClockSource) ?u32 { + return switch (self) { + .xtal => xtal_hz, + .pll_div => pll_div_hz, + .rc_fast => null, + }; + } +}; + +pub const xtal_hz: u32 = 40_000_000; +pub const pll_div_hz: u32 = 80_000_000; +pub const rc_fast_hz_approx: u32 = 17_500_000; + +/// Select the timers' source clock. A read-modify-write of a register that also holds RMT's clock +/// fields, so it runs with interrupts masked, like everything else that touches HP_SYS_CLKRST. +pub fn setClockSource(src: ClockSource) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl22.modify(.{clk_src_sel.is(@intFromEnum(src))}); +} + +pub fn getClockSource() ClockSource { + return @enumFromInt(peri_clk_ctrl22.get(clk_src_sel)); +} + +/// LEDC's core ("function") clock gate. Distinct from the APB gate in `clkrst`: the APB clock makes +/// the registers addressable, this one makes the counters run - and ESP-IDF notes that some LEDC +/// registers and the gamma RAM need it just to be read or written (ledc.c:433-436). +pub fn setFunctionClockEnabled(on: bool) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl22.modify(.{func_clk_en.is(@intFromBool(on))}); +} + +/// Bring the peripheral up, in the only order that works: bus clock, reset, function clock, source. +/// +/// The bus clock first because LEDC is one of the blocks whose APB gate is *off* at power-on +/// (`hp_sys_clkrst_reg.h:835`, REG_LEDC_APB_CLK_EN default 0), so every register read before this +/// returns the last value the bus latched. The function clock before any configuration because the +/// gamma RAM needs it. ESP-IDF deasserts the reset rather than pulsing it (ledc.c:430-431), because +/// its driver may be attaching to a running LEDC; this pulses, which is the stronger guarantee for a +/// fresh boot and is measurably safe on this board - pulsing REG_RST_EN_LEDC for 1 ms left the +/// console untouched and returned LEDC_CH0_CONF0 to 0. +pub fn init(src: ClockSource) void { + clkrst.setClockEnabled(.ledc, true); + clkrst.resetPeripheral(.ledc); + setFunctionClockEnabled(true); + setClockSource(src); +} + +// -------------------------------------------------------------------------------- divider maths + +/// ESP-IDF's `ledc_calculate_divisor`, transcribed from `esp_driver_ledc/src/ledc.c:468-497`: +/// +/// return (((uint64_t) src_clk_freq << LEDC_LL_FRACTIONAL_BITS) + freq_hz * precision / 2) +/// / (freq_hz * precision); +/// +/// Result is Q10.8 - see the file comment - and `divisorValid` says whether it fits the field. +/// +/// Two properties of that C expression are not obvious and are reproduced deliberately, because a +/// HAL that computed a *better* divider than IDF's would disagree with it on real inputs and there +/// would be no way to tell which of the two was wrong: +/// +/// 1. `freq_hz * precision` is `int * uint32_t`, so it is computed in **32 bits and wraps**, and +/// the wrap is not always harmlessly out of range. Ask for 4097 Hz at 20-bit resolution from the +/// 40 MHz XTAL: the true product is 2^32 + 2^20, the C code divides by 2^20 instead, and the +/// answer is 9766 - a *valid* divider, which programs 1.0 Hz. IDF accepts it, because the value +/// passes its own range check. `%*` here is that wrap, on purpose: reproducing it is what makes +/// the on-die comparison meaningful, and the numbers above are how a caller can recognise it. +/// 2. The quotient is `uint64_t` but the return type is `uint32_t`, so it is **truncated**. From a +/// 40 MHz source at 1 Hz and 1-bit resolution the quotient is 5.12e9 and IDF returns 825032704. +/// `@truncate` is that truncation. +/// +/// The one place this cannot follow IDF is `freq_hz * precision == 0`, reachable at exactly 4096 Hz +/// with 20-bit resolution (2^32, wrapping to zero), where the C code divides by zero. Returning 0 is +/// a deliberate substitution: it is not a valid divider, so `divisorValid` rejects it and the caller +/// gets an error instead of undefined behaviour. +pub fn divisor(src_hz: u32, freq_hz: u32, resolution: u5) u32 { + const precision: u32 = @as(u32, 1) << resolution; + const den: u32 = freq_hz *% precision; + if (den == 0) return 0; + const num: u64 = (@as(u64, src_hz) << fractional_bits) + den / 2; + return @truncate(num / den); +} + +/// `LEDC_IS_DIV_INVALID`, inverted (ledc.c:114). A divider below 1.0 means the requested frequency +/// is faster than the source can produce at that resolution. +pub fn divisorValid(div: u32) bool { + return div >= divisor_min and div <= divisor_max; +} + +/// The frequency a given divider and resolution actually produce: `f_src * 256 / (div * 2^res)`, +/// rounded, and 0 for a divider of 0. +/// +/// This is `ledc_get_freq`'s arithmetic (ledc.c:1175) with one deliberate difference: the +/// denominator is computed in 64 bits, so it does not wrap. Nothing compares this against IDF - it +/// is a convenience for callers checking what they got - and a wrapped denominator here would be a +/// bug rather than a compatibility requirement. +pub fn frequencyOf(src_hz: u32, div: u32, resolution: u5) u32 { + if (div == 0) return 0; + const den: u64 = @as(u64, div) * (@as(u64, 1) << resolution); + const num: u64 = (@as(u64, src_hz) << fractional_bits) + den / 2; + return @truncate(num / den); +} + +// --------------------------------------------------------------------------------------- timers + +/// Stage the divider. `ledc_ll_set_clock_divider`, ledc_ll.h:345-348. +pub fn setClockDivider(timer: u32, div: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{clk_div.is(div)}); +} + +pub fn getClockDivider(timer: u32) u32 { + std.debug.assert(timer < timer_count); + return tim_conf.at(timer).get(clk_div); +} + +/// Stage the duty resolution, in bits. `ledc_ll_set_duty_resolution`, ledc_ll.h:391-394. +pub fn setDutyResolution(timer: u32, bits: u5) void { + std.debug.assert(timer < timer_count); + std.debug.assert(bits <= max_duty_resolution); + tim_conf.at(timer).modify(.{duty_res.is(bits)}); +} + +pub fn getDutyResolution(timer: u32) u5 { + std.debug.assert(timer < timer_count); + return @intCast(tim_conf.at(timer).get(duty_res)); +} + +/// Commit the staged divider and resolution. One store, and the bit clears itself. +/// +/// ESP-IDF does not wait for it: "we don't wait for the bit gets cleared since it can take quite +/// long depends on the pwm frequency" (ledc_ll.h:289). Neither does this - a poll here would block +/// for a whole PWM period, and there is nothing useful to do with the answer. +pub fn commitTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{tim_para_up.is(1)}); +} + +/// Reset the timer's counter: assert, deassert (`ledc_ll_timer_rst`, ledc_ll.h:301-305). +/// +/// Note the reset value of `LEDC_TIMERn_RST` is **1** (ledc_reg.h:936-943), which is one of the +/// 46.7% of fields whose reset value is not zero, and the reason `configureTimer` finishes by +/// clearing it: a freshly reset LEDC block holds all four counters at zero and they stay there until +/// something writes that bit back down. +pub fn resetTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + const r = tim_conf.at(timer); + r.modify(.{tim_rst.is(1)}); + r.modify(.{tim_rst.is(0)}); +} + +/// Freeze the counter where it is (`ledc_ll_timer_pause`, ledc_ll.h:316-319). +pub fn pauseTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{tim_pause.is(1)}); +} + +pub fn resumeTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{tim_pause.is(0)}); +} + +/// The counter's current value, 20 bits. Reading it is a plain load - no latch handshake, unlike +/// systimer. +pub fn timerCount(timer: u32) u32 { + std.debug.assert(timer < timer_count); + return tim_value.at(timer).raw(); +} + +/// Everything a timer needs to produce `freq_hz` at `resolution` bits, in ESP-IDF's order: +/// divider, resolution, commit, then out of pause and out of reset (`ledc_set_timer_params`, +/// ledc.c:244-261, followed by ledc.c:816-818). +/// +/// Returns `DividerOutOfRange` rather than programming a divider the hardware cannot hold. The +/// caller passes the source frequency because this HAL has no clock tree: `ClockSource.hz()` gives +/// it for XTAL and PLL_DIV, and RC_FAST has to be measured. +pub fn configureTimer(timer: u32, opts: struct { + src_hz: u32, + freq_hz: u32, + resolution: u5, +}) Error!void { + std.debug.assert(timer < timer_count); + if (opts.resolution == 0 or opts.resolution > max_duty_resolution) return Error.DutyResolutionOutOfRange; + const div = divisor(opts.src_hz, opts.freq_hz, opts.resolution); + if (!divisorValid(div)) return Error.DividerOutOfRange; + + setClockDivider(timer, div); + setDutyResolution(timer, opts.resolution); + commitTimer(timer); + resumeTimer(timer); + resetTimer(timer); +} + +// ------------------------------------------------------------------------------------- channels + +/// Which timer drives this channel. Staged; needs `commitChannel`. +/// `ledc_ll_bind_channel_timer`, ledc_ll.h:697-700. +pub fn bindTimer(channel: u32, timer: u32) void { + std.debug.assert(channel < channel_count and timer < timer_count); + ch_conf0.at(channel).modify(.{timer_sel.is(timer)}); +} + +pub fn boundTimer(channel: u32) u32 { + std.debug.assert(channel < channel_count); + return ch_conf0.at(channel).get(timer_sel); +} + +/// Where in the period the output goes high, in counter ticks. Staged. +/// `ledc_ll_set_hpoint`, ledc_ll.h:450-453. +pub fn setHpoint(channel: u32, value: u32) void { + std.debug.assert(channel < channel_count and value <= hpoint.max()); + ch_hpoint.at(channel).modify(.{hpoint.is(value)}); +} + +/// Stage a duty value, in counter ticks out of `2^resolution`. +/// +/// Two things happen here that the name does not suggest, and both are ESP-IDF's +/// (`ledc_ll_set_duty_int_part` ledc_ll.h:480-483, `ledc_duty_config` ledc.c:263-280): +/// +/// * The register holds duty in **Q21.4** - four fractional bits, used by fades - so the integer +/// duty is shifted left by 4. `getDuty` shifts back. +/// * The P4 has no plain-duty path. `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` moved out of +/// `CHn_CONF1` into gamma RAM, so a constant duty is a one-step fade of scale 0: entry 0 gets +/// (increase, one cycle, scale 0, one step) and the range count is set to 1. Without that entry +/// the staged duty is committed and the output does not move. +/// +/// Staged; needs `commitChannel` (or `start`, which commits). +pub fn setDuty(channel: u32, value: u32) void { + std.debug.assert(channel < channel_count); + std.debug.assert(value <= duty.max() >> 4); + ch_duty.at(channel).modify(.{duty.is(value << 4)}); + stageNoFade(channel); +} + +/// The duty the hardware is currently using, from the read-only shadow (`ledc_ll_get_duty`, +/// ledc_ll.h:495-498). This is the one register that shows whether a commit actually happened - and +/// it only updates when the timer next overflows, so it is not a synchronous read-back. +pub fn currentDuty(channel: u32) u32 { + std.debug.assert(channel < channel_count); + return ch_duty_r.at(channel).get(duty_r) >> 4; +} + +/// Gamma RAM entry 0 as "no fade": one step, one cycle, scale 0, increasing. Exactly the parameters +/// `ledc_set_duty` passes down (ledc.c:1109-1117) for a constant duty. +fn stageNoFade(channel: u32) void { + // The whole word is being established, and every field in it is being named, so this is one of + // the few places `write` is right rather than `modify`. + ch_gamma_range0.at(channel).write(.{ + gamma_duty_inc.is(1), + gamma_duty_cycle.is(1), + gamma_scale.is(0), + gamma_duty_num.is(1), + }); + ch_gamma_conf.at(channel).modify(.{gamma_entry_num.is(1)}); +} + +/// The output driver. Staged; needs `commitChannel`. +/// `ledc_ll_set_sig_out_en`, ledc_ll.h:592-596. +pub fn setOutputEnabled(channel: u32, on: bool) void { + std.debug.assert(channel < channel_count); + ch_conf0.at(channel).modify(.{sig_out_en.is(@intFromBool(on))}); +} + +/// The level the pad holds while the channel is disabled - and only while it is disabled +/// (`ledc_reg.h:34-37`: "Valid only when LEDC_SIG_OUT_EN_CHn is 0"). Staged. +/// `ledc_ll_set_idle_level`, ledc_ll.h:622-626. +pub fn setIdleLevel(channel: u32, level: u1) void { + std.debug.assert(channel < channel_count); + ch_conf0.at(channel).modify(.{idle_lv.is(level)}); +} + +/// Hand the staged duty to the fade engine. `ledc_ll_set_duty_start`, ledc_ll.h:607-610. +/// +/// `DUTY_START` lives in `CHn_CONF1`, alone, and is annotated `R/W/SC` - the hardware clears it when +/// the (here one-step) fade finishes. A read-modify-write is still the right store: the bit is the +/// only field in the word, but bits 30:0 are reserved and writing them back as read is what IDF's +/// bitfield assignment does. +pub fn startFade(channel: u32) void { + std.debug.assert(channel < channel_count); + ch_conf1.at(channel).modify(.{duty_start.is(1)}); +} + +/// Commit the channel's staged fields: `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV`, `HPOINT`, +/// `DUTY_START`, `OVF_CNT_EN` and the duty (`ledc_reg.h:42-47`). +/// +/// One deliberate store, never folded into the store that staged the values, matching +/// `ledc_ll_ls_channel_update` (ledc_ll.h:435-438). It is a read-modify-write because the commit bit +/// shares its word with the staged fields - see the file comment - and that is safe only because the +/// bit reads back as 0. +pub fn commitChannel(channel: u32) void { + std.debug.assert(channel < channel_count); + ch_conf0.at(channel).modify(.{ch_para_up.is(1)}); +} + +/// Start driving: output on, duty handed over, committed. `_ledc_update_duty`, ledc.c:1021-1026. +pub fn start(channel: u32) void { + setOutputEnabled(channel, true); + startFade(channel); + commitChannel(channel); +} + +/// Stop driving and hold the pad at `idle_level`. `ledc_stop`, ledc.c:1039-1050. +/// +/// The order is IDF's and it matters: the idle level is staged *before* the output is disabled, so +/// the two reach the hardware in the same commit and the pad never spends a period at the old idle +/// level. +pub fn stop(channel: u32, idle_level: u1) void { + setIdleLevel(channel, idle_level); + setOutputEnabled(channel, false); + commitChannel(channel); +} + +/// A whole channel in one commit: timer, duty, hpoint, idle level, output enable. +/// +/// This is the one operation here that is not a transcription of an ESP-IDF function - IDF's +/// `ledc_channel_config` also allocates a driver object, reserves the pin and installs a fade +/// service - but it is the same register sequence: stage everything, then commit once. One commit +/// rather than five is the point: the channel changes all at once, at a period boundary, instead of +/// drifting through four intermediate configurations. +pub fn configureChannel(channel: u32, opts: struct { + timer: u32, + duty: u32, + hpoint: u32 = 0, + idle_level: u1 = 0, + output_enabled: bool = true, +}) void { + bindTimer(channel, opts.timer); + setHpoint(channel, opts.hpoint); + setDuty(channel, opts.duty); + setIdleLevel(channel, opts.idle_level); + setOutputEnabled(channel, opts.output_enabled); + startFade(channel); + commitChannel(channel); +} + +// ------------------------------------------------------------------------------------ pin output + +/// The GPIO matrix signal index for a channel's output. `ledc_periph_signal[0].sig_out0_idx` is +/// `LEDC_LS_SIG_OUT_PAD_OUT0_IDX` (esp_hal_ledc/esp32p4/ledc_periph.c:14-18) and the driver adds the +/// channel number to it (ledc.c:831); the eight indices are consecutive from 126, asserted above. +pub fn signalIndex(channel: u32) u32 { + std.debug.assert(channel < channel_count); + return @as(u32, @intCast(regs.LEDC_LS_SIG_OUT_PAD_OUT0_IDX)) + channel; +} + +/// Route a channel's output to a pad through the GPIO matrix. No LEDC register is involved: the +/// peripheral has no pad of its own, and this is the whole of `ledc_set_pin`'s hardware effect +/// (ledc.c:823-836, whose `gpio_matrix_output` is func_sel + matrix source + output-enable control, +/// gpio_hal.c:60-69). +pub fn attachPin(channel: u32, pin: u8) void { + gpio.matrixOut(pin, signalIndex(channel)); +} + +// ----------------------------------------------------------------------------------------- tests + +test "the divider is Q10.8: integer part in [17:8], fraction in [7:0]" { + // 40 MHz XTAL, 1 kHz, 13-bit resolution. 40e6*256/(1000*8192) = 1250 = 0x4E2, i.e. 4 + 226/256 + // = 4.8828. Checked against ESP-IDF's own expression compiled on the host over a 1,680-point + // sweep of (source, frequency, resolution). + try std.testing.expectEqual(@as(u32, 1250), divisor(40_000_000, 1_000, 13)); + try std.testing.expectEqual(@as(u32, 1250 >> 8), 4); + try std.testing.expectEqual(@as(u32, 1250 & 0xff), 226); + // And back again, to within the rounding the format allows. + try std.testing.expectEqual(@as(u32, 1_000), frequencyOf(40_000_000, 1250, 13)); +} + +test "divider values for the frequencies the differential harness uses" { + try std.testing.expectEqual(@as(u32, 2000), divisor(40_000_000, 5_000, 10)); + try std.testing.expectEqual(@as(u32, 500), divisor(40_000_000, 20_000, 10)); + try std.testing.expectEqual(@as(u32, 2083), divisor(40_000_000, 300, 14)); + // 80 MHz PLL_F80M, same request: exactly twice the divider. + try std.testing.expectEqual(@as(u32, 4000), divisor(80_000_000, 5_000, 10)); +} + +test "the arithmetic reproduces IDF's overflow and truncation rather than fixing them" { + // 32-bit wrap of freq*precision: the true product at 1 MHz / 13 bits is 8_192_000_000, and the + // C expression divides by 3_897_032_704 instead, giving 3 where the unwrapped arithmetic would + // give 1. Neither is a usable divider - both are below 1.0, so `divisorValid` rejects them the + // way `LEDC_IS_DIV_INVALID` does - but the *value* has to be IDF's, or a caller comparing the + // two implementations sees a difference that is really just two different roundings. + try std.testing.expectEqual(@as(u32, 1_000_000 *% (@as(u32, 1) << 13)), 3_897_032_704); + try std.testing.expectEqual(@as(u32, 3), divisor(40_000_000, 1_000_000, 13)); + try std.testing.expect(!divisorValid(divisor(40_000_000, 1_000_000, 13))); + // u64 quotient truncated to u32, exactly as the C return type does. + try std.testing.expectEqual(@as(u32, 825_032_704), divisor(40_000_000, 1, 1)); + // The one input where IDF divides by zero: 4096 * 2^20 == 2^32. + try std.testing.expectEqual(@as(u32, 0), divisor(40_000_000, 4096, 20)); + try std.testing.expect(!divisorValid(divisor(40_000_000, 4096, 20))); + // The wrap that is *not* self-limiting: 4097 Hz at 20 bits gives a divider IDF's own range check + // accepts, and it programs 1.0 Hz. Reproduced rather than corrected, because the point of the + // differential test is to be wrong in the same way IDF is or not at all. + try std.testing.expectEqual(@as(u32, 9766), divisor(40_000_000, 4097, 20)); + try std.testing.expect(divisorValid(9766)); + try std.testing.expectEqual(@as(u32, 1), frequencyOf(40_000_000, 9766, 20)); +} + +test "validity is the field's range, not the whole u32" { + try std.testing.expect(!divisorValid(0xff)); // below 1.0 + try std.testing.expect(divisorValid(0x100)); // exactly 1.0 + try std.testing.expect(divisorValid(0x3ffff)); + try std.testing.expect(!divisorValid(0x40000)); + // 40 MHz cannot make 5 kHz at 13 bits: that needs a divider of 0.98. + try std.testing.expect(!divisorValid(divisor(40_000_000, 5_000, 13))); +} diff --git a/src/hal/rwdt.zig b/src/hal/rwdt.zig new file mode 100644 index 0000000..5002400 --- /dev/null +++ b/src/hal/rwdt.zig @@ -0,0 +1,123 @@ +//! The RTC watchdog (RWDT) and the super watchdog (SWD), in the always-on LP domain. +//! +//! This module exists because of a bug that had been in every application in this project since the +//! first one, and was invisible for a simple reason: nothing had ever run for more than eight +//! seconds. +//! +//! The second-stage bootloader arms the RTC watchdog to cover the handover to the application, and +//! expects the application to take it over - ESP-IDF disables it in `esp_system`'s startup, which a +//! bare image never runs. So the board resets, and the console says so if anyone looks: +//! +//! MARK ZIG_P4_ALIVE beat=8 gpio20 high=1 low=0 +//! rst:0x10 (CHIP_LP_WDT_RESET),boot:0x30f (SPI_FAST_FLASH_BOOT) +//! +//! Every demo, every example and every hardware test in this repo had been silently rebooting on a +//! roughly ten-second cycle. It surfaced only when the differential harness grew past 26 cases and +//! the run stopped fitting inside one watchdog period - which first looked like "the UART suite +//! crashes the board", and was not. +//! +//! Two watchdogs live here and both have to be dealt with: +//! +//! * **RWDT**, the RTC watchdog proper, in `LP_WDT_CONFIG0_REG`. Write-protected. +//! * **SWD**, the super watchdog, a separate always-on timer whose job is to catch a system that +//! has stopped feeding everything else. It has its own key and its own register, and the +//! bootloader leaves it auto-feeding (`bootloader_super_wdt_auto_feed`); an application that +//! disables RWDT and forgets SWD gets a longer fuse rather than no fuse. +//! +//! The write-protect scheme is the same as the timer groups': the key register's *reset value* is +//! the unlock key, so writing anything else locks it. Writes to a locked register are dropped +//! silently - no fault, no status bit - which is why `disable()` verifies afterwards and returns +//! whether it took. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +const config0 = Reg.at(regs.LP_WDT_CONFIG0_REG); +const wprotect = Reg.at(regs.LP_WDT_WPROTECT_REG); +const swd_config = Reg.at(regs.LP_WDT_SWD_CONFIG_REG); +const swd_wprotect = Reg.at(regs.LP_WDT_SWD_WPROTECT_REG); + +const wdt_en = Field.of(regs.LP_WDT_WDT_EN_S, regs.LP_WDT_WDT_EN_V); +/// Flash-boot mode runs the watchdog independently of `wdt_en`, which is how the bootloader keeps +/// the fuse lit across the handover. Clearing `wdt_en` alone leaves this armed. +const flashboot_en = Field.of(regs.LP_WDT_WDT_FLASHBOOT_MOD_EN_S, regs.LP_WDT_WDT_FLASHBOOT_MOD_EN_V); +const swd_disable = Field.of(regs.LP_WDT_SWD_DISABLE_S, regs.LP_WDT_SWD_DISABLE_V); +const swd_auto_feed = Field.of(regs.LP_WDT_SWD_AUTO_FEED_EN_S, regs.LP_WDT_SWD_AUTO_FEED_EN_V); +const swd_feed = Field.of(regs.LP_WDT_SWD_FEED_S, regs.LP_WDT_SWD_FEED_V); +const feed_reg = Reg.at(regs.LP_WDT_FEED_REG); +const feed_bit = Field.of(regs.LP_WDT_FEED_S, regs.LP_WDT_FEED_V); + +/// The unlock key for both blocks, which is also each key register's reset value: "if the register +/// contains a different value than its reset value, write protection is enabled" +/// (lp_wdt_reg.h). `LP_WDT_WKEY_VALUE` and `LP_WDT_SWD_WKEY_VALUE` in +/// esp_hal_wdt/esp32p4/include/hal/lpwdt_ll.h:25,27 are both this number. +const wkey: u32 = 0x50D8_3AA1; + +/// Unlocked access to the RTC watchdog. `defer guard.release()` re-locks. +pub const Guard = struct { + pub inline fn release(_: Guard) void { + // Anything that is not the key locks it. ESP-IDF writes 0 (lpwdt_ll.h), so this does too: + // it keeps the register comparable against IDF's in a differential test. + wprotect.writeRaw(0); + } +}; + +pub inline fn unlock() Guard { + wprotect.writeRaw(wkey); + return .{}; +} + +/// Turn the RTC watchdog off, and stop the super watchdog behind it. +/// +/// Returns false if the write did not take, which means the key was wrong: a protected register +/// swallows writes without complaint, so the only way to know is to read back. +pub fn disable() bool { + { + const guard = unlock(); + defer guard.release(); + // Both bits, in one store: clearing `wdt_en` while leaving flash-boot mode armed is the + // half-fix that still reboots. + config0.modify(.{ wdt_en.is(0), flashboot_en.is(0) }); + } + + // The super watchdog is a separate block with its own key. + swd_wprotect.writeRaw(wkey); + swd_config.modify(.{ swd_disable.is(1), swd_auto_feed.is(0) }); + swd_wprotect.writeRaw(0); + + return config0.get(wdt_en) == 0 and config0.get(flashboot_en) == 0 and swd_config.get(swd_disable) == 1; +} + +/// Feed the RTC watchdog instead of disabling it, for an application that would rather keep the +/// protection. +/// +/// The counter is fed through its own register, `LP_WDT_FEED_REG`, not through anything in +/// CONFIG0 - ESP-IDF's `lpwdt_ll_feed` writes `hw->feed.feed = 1`. An earlier version of this +/// function read a CONFIG0 field and wrote the same value back, which is a pure no-op: the register +/// ended with the bits it started with and the counter kept running. An application that took this +/// module's own advice - keep the protection, feed it - would have been reset about ten seconds +/// later with nothing on the console, which is the exact failure this file exists to document. The +/// differential harness could not have caught it either, because a no-op leaves the register +/// bit-identical. +pub fn feed() void { + const guard = unlock(); + defer guard.release(); + feed_reg.write(.{feed_bit.is(1)}); +} + +/// Feed the super watchdog once. Independent of RWDT and of its own auto-feed setting. +pub fn feedSuper() void { + swd_wprotect.writeRaw(wkey); + swd_config.modify(.{swd_feed.is(1)}); + swd_wprotect.writeRaw(0); +} + +/// Whether either watchdog is still armed - worth printing once at startup, because the symptom of +/// getting this wrong is a reset ten seconds later with no other clue. +pub fn armed() bool { + return config0.get(wdt_en) == 1 or config0.get(flashboot_en) == 1 or swd_config.get(swd_disable) == 0; +} diff --git a/src/hal/sdmmc.zig b/src/hal/sdmmc.zig new file mode 100644 index 0000000..0bebeb0 --- /dev/null +++ b/src/hal/sdmmc.zig @@ -0,0 +1,2002 @@ +//! The SDMMC host controller, driven as an **SDIO host**. +//! +//! There is no SD card on this board. Slot 1 of the P4's SDMMC controller goes to an ESP32-C6 +//! running ESP-Hosted coprocessor firmware, which presents itself as a 4-bit SDIO device: CLK 18, +//! CMD 19, D0-D3 = 14/15/16/17. So this file implements CMD0/CMD5/CMD3/CMD7 and then CMD52/CMD53, +//! and nothing above them. SD memory cards, SPI mode, CSD/CID decoding and block devices are +//! deliberately absent - they are a different problem that happens to share a peripheral. +//! +//! The controller is a Synopsys DesignWare mobile-storage host. Three of its properties decide the +//! shape of everything below. +//! +//! **The card clock is not the register clock.** CLKDIV, CLKSRC and CLKENA are written on the bus +//! side and do not reach the card-interface unit until a *clock update command* is issued: a write +//! to the CMD register with `update_clk_reg` and `start_command` set, which sends nothing to the +//! card (`sdmmc_reg.h:440-456`, and ESP-IDF's `sd_host_slot_clock_update_command`, +//! `sd_host_sdmmc.c:896-912`). A driver that programmes a divider and moves on has changed +//! nothing. Three such commands are needed to change frequency safely - clock off, reprogramme, +//! clock on - and that is what `setBusClock` does. +//! +//! **The command register is a single word, and `start_command` is bit 31 of it.** Every attribute +//! of a command - index, whether a response is expected, whether its CRC is checked, whether data +//! follows and in which direction - is a field of the same word, and writing that word with bit 31 +//! set launches the command. So the interesting part of "send CMD52" is an encoding, not a +//! sequence, and `commandWord` is a pure function of the request. It is host-tested, and the +//! oracle compares the words it produces against words built through ESP-IDF's own +//! `sdmmc_hw_cmd_t` bitfields. +//! +//! **Data moves by internal DMA over descriptors in memory, and the P4 caches that memory.** +//! `soc_caps.h:185` sets SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE, so L2MEM - where every static in this +//! image lives - is reached by the CPU through the L1 data cache while the IDMAC reaches it +//! directly. See the "Cache" section below for the resolution; it is the one place in this file +//! where the right answer is not visible in any register header. +//! +//! Nothing here has been run on hardware by the author of this file. What is claimed is that the +//! register arithmetic and the command encodings match ESP-IDF's at the cited lines, that +//! `src/oracle/sdmmc_cases.zig` compares the two on the die, and that the configuration `init` +//! leaves behind reproduces a dump taken from a working ESP-IDF image on this board. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); +const intr = @import("intr.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +pub const Error = error{ Timeout, CrcError, ResponseError, NotSupported, Busy }; + +// ------------------------------------------------------------------------------- registers +// +// One instance, at DR_REG_SDHOST_BASE = DR_REG_SDMMC_BASE = 0x50083000 (`reg_base.h:44`, `:204`, +// and `esp32p4.peripherals.ld:41` agrees). The macros are spelled SDHOST_*, the peripheral is +// spelled SDMMC, and both names are ESP-IDF's. + +const ctrl = Reg.at(regs.SDHOST_CTRL_REG); +const clkdiv = Reg.at(regs.SDHOST_CLKDIV_REG); +const clksrc = Reg.at(regs.SDHOST_CLKSRC_REG); +const clkena = Reg.at(regs.SDHOST_CLKENA_REG); +const tmout = Reg.at(regs.SDHOST_TMOUT_REG); +const ctype = Reg.at(regs.SDHOST_CTYPE_REG); +const blksiz = Reg.at(regs.SDHOST_BLKSIZ_REG); +const bytcnt = Reg.at(regs.SDHOST_BYTCNT_REG); +const intmask = Reg.at(regs.SDHOST_INTMASK_REG); +const cmdarg = Reg.at(regs.SDHOST_CMDARG_REG); +const cmd = Reg.at(regs.SDHOST_CMD_REG); +const resp0 = Reg.at(regs.SDHOST_RESP0_REG); +const rintsts = Reg.at(regs.SDHOST_RINTSTS_REG); +/// The *masked* status: RINTSTS gated by INTMASK, and the only word the controller's interrupt +/// output looks at. ESP-IDF's `sdmmc_ll_get_intr_status` reads this one and not RINTSTS +/// (`sdmmc_ll.h:841-844`), which is exactly why INTMASK decides what reaches the CLIC while +/// RINTSTS stays readable for the polling path. +const mintsts = Reg.at(regs.SDHOST_MINTSTS_REG); +const status = Reg.at(regs.SDHOST_STATUS_REG); +const fifoth = Reg.at(regs.SDHOST_FIFOTH_REG); +const bmod = Reg.at(regs.SDHOST_BMOD_REG); +const pldmnd = Reg.at(regs.SDHOST_PLDMND_REG); +const dbaddr = Reg.at(regs.SDHOST_DBADDR_REG); +const idsts = Reg.at(regs.SDHOST_IDSTS_REG); +const idinten = Reg.at(regs.SDHOST_IDINTEN_REG); + +// CTRL fields. Two of them - `dma_enable` at bit 5 and `use_internal_dma` at bit 25 - have no +// `_S`/`_V` macro pair in `sdmmc_reg.h` at all: that header documents CTRL as bits 0,1,2,4,6..11 +// and stops. They are real, they are in the measured working dump (`ctrl=0x02000030`), and +// `sdmmc_struct.h:76` and `:135` name them at exactly those positions. This is the same situation +// as the IO MUX pull bits in `hal/gpio.zig`, and the same remedy: `Field.bit` with the struct +// header cited, because the struct header is ESP-IDF's definition of the layout even where the +// macro header is incomplete. +const controller_reset = Field.of(regs.SDHOST_CONTROLLER_RESET_S, regs.SDHOST_CONTROLLER_RESET_V); +const fifo_reset = Field.of(regs.SDHOST_FIFO_RESET_S, regs.SDHOST_FIFO_RESET_V); +const dma_reset = Field.of(regs.SDHOST_DMA_RESET_S, regs.SDHOST_DMA_RESET_V); +const int_enable = Field.of(regs.SDHOST_INT_ENABLE_S, regs.SDHOST_INT_ENABLE_V); +/// `sdmmc_struct.h:76` - `uint32_t dma_enable:1;` immediately after `int_enable:1` at bit 4. +const dma_enable = Field.bit(5); +/// `sdmmc_struct.h:135` - after `reserved2:4`, `card_voltage_a:4`, `card_voltage_b:4` and +/// `enable_od_pullup:1`, i.e. bit 25. `sdmmc_ll_enable_dma` (`sdmmc_ll.h:812-818`) is the only +/// writer, and the working dump's `ctrl=0x02000030` has exactly this bit plus 4 and 5. +const use_internal_dma = Field.bit(25); + +const clk_divider0 = Field.of(regs.SDHOST_CLK_DIVIDER0_S, regs.SDHOST_CLK_DIVIDER0_V); +const clk_divider1 = Field.of(regs.SDHOST_CLK_DIVIDER1_S, regs.SDHOST_CLK_DIVIDER1_V); +// CLKSRC is documented as one 4-bit field, two bits per card ("bit[1:0] are assigned for card 0, +// bit[3:2] are assigned for card 1", `sdmmc_reg.h:166-179`). `sdmmc_struct.h:191-192` splits it +// into `card0:2` and `card1:2`, which is the shape a driver wants; there are no macros for the +// halves, so the two sub-fields are spelled out with that citation. +const clksrc_card0 = Field.of(0, 0x3); +const clksrc_card1 = Field.of(2, 0x3); +const cclk_enable = Field.of(regs.SDHOST_CCLK_ENABLE_S, regs.SDHOST_CCLK_ENABLE_V); +const lp_enable = Field.of(regs.SDHOST_LP_ENABLE_S, regs.SDHOST_LP_ENABLE_V); +const response_timeout = Field.of(regs.SDHOST_RESPONSE_TIMEOUT_S, regs.SDHOST_RESPONSE_TIMEOUT_V); +const data_timeout = Field.of(regs.SDHOST_DATA_TIMEOUT_S, regs.SDHOST_DATA_TIMEOUT_V); +const card_width4 = Field.of(regs.SDHOST_CARD_WIDTH4_S, regs.SDHOST_CARD_WIDTH4_V); +const card_width8 = Field.of(regs.SDHOST_CARD_WIDTH8_S, regs.SDHOST_CARD_WIDTH8_V); +const block_size = Field.of(regs.SDHOST_BLOCK_SIZE_S, regs.SDHOST_BLOCK_SIZE_V); +const byte_count = Field.of(regs.SDHOST_BYTE_COUNT_S, regs.SDHOST_BYTE_COUNT_V); +const int_mask = Field.of(regs.SDHOST_INT_MASK_S, regs.SDHOST_INT_MASK_V); +const sdio_int_mask = Field.of(regs.SDHOST_SDIO_INT_MASK_S, regs.SDHOST_SDIO_INT_MASK_V); +const data_busy = Field.of(regs.SDHOST_DATA_BUSY_S, regs.SDHOST_DATA_BUSY_V); +const tx_wmark = Field.of(regs.SDHOST_TX_WMARK_S, regs.SDHOST_TX_WMARK_V); +const rx_wmark = Field.of(regs.SDHOST_RX_WMARK_S, regs.SDHOST_RX_WMARK_V); +const dma_msize = Field.of(regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_S, regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_V); +const bmod_swr = Field.of(regs.SDHOST_BMOD_SWR_S, regs.SDHOST_BMOD_SWR_V); +const bmod_fb = Field.of(regs.SDHOST_BMOD_FB_S, regs.SDHOST_BMOD_FB_V); +const bmod_de = Field.of(regs.SDHOST_BMOD_DE_S, regs.SDHOST_BMOD_DE_V); +const idinten_ti = Field.of(regs.SDHOST_IDINTEN_TI_S, regs.SDHOST_IDINTEN_TI_V); +const idinten_ri = Field.of(regs.SDHOST_IDINTEN_RI_S, regs.SDHOST_IDINTEN_RI_V); +const idinten_ni = Field.of(regs.SDHOST_IDINTEN_NI_S, regs.SDHOST_IDINTEN_NI_V); + +// The host-side clock generator, which is *not* in the SDMMC block: the P4 moved it into +// HP_SYS_CLKRST, and it is the first of two divider stages (this one, then CLKDIV inside the +// controller). `sdmmc_ll.h:227-228` for the source mux and gate, `:244-258` for the divider, +// `:305-315` for the sampling/driving phase clocks. +const peri_clk_ctrl01 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL01_REG); +const peri_clk_ctrl02 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL02_REG); + +const sdio_hs_mode = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_S, regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_V); +const sdio_ls_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_V); +const sdio_ls_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_V); +const sdio_ls_clk_edge_cfg_update = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_V); +const sdio_ls_clk_edge_l = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_V); +const sdio_ls_clk_edge_h = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_V); +const sdio_ls_clk_edge_n = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_V); +const sdio_ls_slf_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_V); +const sdio_ls_drv_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_V); +const sdio_ls_sam_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_V); +const sdio_ls_slf_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_V); +const sdio_ls_drv_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_V); +const sdio_ls_sam_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_V); + +// --------------------------------------------------------------------------------- interrupts +// +// RINTSTS / INTMASK share one 16-bit layout plus a 2-bit per-card SDIO field at [17:16]. +// `sdmmc_ll.h:35-53` names every bit; the numbers below are those, not a re-derivation. + +pub const Event = struct { + pub const cd: u32 = 1 << 0; // card detect + pub const re: u32 = 1 << 1; // response error + pub const cmd_done: u32 = 1 << 2; + pub const dto: u32 = 1 << 3; // data transfer over + pub const txdr: u32 = 1 << 4; + pub const rxdr: u32 = 1 << 5; + pub const rcrc: u32 = 1 << 6; // response CRC error + pub const dcrc: u32 = 1 << 7; // data CRC error + pub const rto: u32 = 1 << 8; // response timeout + pub const drto: u32 = 1 << 9; // data read timeout + pub const hto: u32 = 1 << 10; // data starvation by host timeout + pub const frun: u32 = 1 << 11; // FIFO under/overrun + pub const hle: u32 = 1 << 12; // hardware locked write error + pub const sbe: u32 = 1 << 13; // RX start-bit error + pub const acd: u32 = 1 << 14; // auto command done + pub const ebe: u32 = 1 << 15; // end-bit error + pub const io_slot0: u32 = 1 << 16; + pub const io_slot1: u32 = 1 << 17; + + /// What `sdmmc_ll.h:64-69` (SDMMC_LL_EVENT_DEFAULT) enables at init. Kept exactly as ESP-IDF + /// spells it, because the oracle compares against it; what this driver actually unmasks is + /// `armed`, below. + pub const default: u32 = cd | re | cmd_done | dto | rcrc | dcrc | rto | drto | hto | hle | sbe | ebe; + + /// `default` without card detect, and the only mask `configureInterrupts` ever writes. + /// + /// Bit 0 has to go, and this is not a preference. There is no card-detect pin on this board: + /// `configurePins` ties the signal to a matrix constant 0 ("card present"), and the + /// transition it makes while doing so *latches* RINTSTS.cd. RINTSTS is a sticky + /// write-1-to-clear register and nothing in the command path clears bit 0 - `sendCommand` + /// deliberately writes `default & ~cd` so as not to disturb asynchronous events. So with cd + /// unmasked, the controller's single output line into the CLIC is asserted from bring-up + /// onwards and never deasserts, and anyone who enables that CLIC line takes an interrupt + /// storm that no handler can end. Found by RxPath on CLIC line 21; the fix belongs here + /// rather than in the handler, because a level output that nothing can lower is this file's + /// bug. + /// + /// The two SDIO card-interrupt bits are absent from both masks: `setSlaveInterruptEnabled` + /// turns the one for this slot on when somebody is prepared to service it. + pub const armed: u32 = default & ~cd; + + /// Anything in here means the command failed. `sdmmc_ll.h:71-77` calls the superset + /// SDMMC_LL_SD_EVENT_MASK; this is the error half of it. + pub const command_errors: u32 = re | rcrc | rto | hle; + pub const data_errors: u32 = dcrc | drto | hto | frun | sbe | ebe; +}; + +/// The IDMAC's five reportable events - TI, RI, FBE, DU, CES - as one mask. `sdmmc_ll.h:83` +/// SDMMC_LL_EVENT_DMA_MASK. +const idsts_event_mask: u32 = 0x1f; + +/// The CLIC source this controller raises, for a caller that wants to be woken rather than to +/// poll. Registering a handler is `hal.intr`'s job and not this file's: see the note on +/// `slaveInterruptPending`. +pub const interrupt_source = intr.Source.sdio_host; + +// -------------------------------------------------------------------------------------- cache +// +// The IDMAC reads its descriptors and its data buffer straight out of L2MEM. The CPU reaches the +// same L2MEM through the L1 data cache (`soc_caps.h:185`, SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE), and +// that cache is write-back: `esp_cache_msync(..., DIR_C2M)` exists precisely because a store the +// CPU has made may still be sitting in a dirty line when the DMA engine reads memory. +// +// ESP-IDF offers two ways out and uses both. `sd_trans_sdmmc.c:135-139` writes descriptors through +// the normal address and calls `esp_cache_msync` after every one. `gdma_link.c:100-118` does it +// the other way: one write-back-and-invalidate when the region is created, and from then on every +// CPU access goes through the non-cacheable alias at `addr + 0x40000000` +// (`hal/cache_ll.h:27` CACHE_LL_L2MEM_NON_CACHE_ADDR, `soc/ext_mem_defs.h:68`). +// +// **This file takes the second route.** It is the cheaper one - no cache call in the transfer +// path - and it is the only one that stays correct without a cache HAL this project does not have. +// The one-time write-back-and-invalidate is still required, and skipping it is a real bug rather +// than a theoretical one: `_start` clears .bss with ordinary stores (`src/main.zig:85-92`), so +// every word of the DMA region below starts life as a *dirty* cache line full of zeros. Nothing +// says when those lines are evicted; if one is written back after a descriptor has been prepared +// through the alias, the descriptor becomes zero and the IDMAC stalls on an unowned descriptor. +// `gdma_link.c:107-112` does exactly this call for exactly this reason. +// +// The two ROM entry points are addressed directly rather than declared `extern`, because the +// generated linker script provides only `ets_printf` and `ets_delay_us`. The addresses are +// ESP-IDF's, from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld:186` and `:190` - the hw_ver1 +// file, which is the one that matches this die. (If they move into the linker script beside the +// other two, these two lines become `extern fn` and nothing else changes.) + +/// `soc/ext_mem_defs.h:68` SOC_NON_CACHEABLE_OFFSET. +pub const non_cacheable_offset: u32 = 0x4000_0000; + +/// `cache_ll_l1_dcache_get_line_size` reports this on the P4, and `sdmmc_struct.h:36-38` states it +/// in prose: "On P4, L1 Cache alignment is 64B". +pub const cache_line: u32 = 64; + +/// `rom/cache.h:230` - CACHE_MAP_L1_DCACHE is BIT(4). +const cache_map_l1_dcache: u32 = 1 << 4; + +const romCacheWriteBackAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int = + @ptrFromInt(0x4fc0_03f4); +const romCacheInvalidateAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int = + @ptrFromInt(0x4fc0_03e4); + +// ------------------------------------------------------------------------------- DMA descriptor + +/// One IDMAC descriptor, exactly as the hardware reads it: `sdmmc_struct.h:13-41`. +/// +/// ESP-IDF's `sdmmc_desc_t` is 64 bytes, not 16, and its own comment says why and when not to: +/// "These `reserved[12]` are for cache alignment... For those who want to access the DMA +/// descriptor in a non-cacheable way, you can consider remove these `reserved[12]` bytes" +/// (`sdmmc_struct.h:35-39`). That is this file, so the padding is gone and the descriptor is the +/// 16 bytes the IDMAC actually fetches. +pub const Descriptor = extern struct { + flags: u32, + /// [12:0] buffer1_size, [25:13] buffer2_size. + sizes: u32, + buffer1: u32, + /// Also `buffer2_ptr`; which one it is depends on `second_address_chained`. + next: u32, + + pub const disable_int_on_completion: u32 = 1 << 1; + pub const last_descriptor: u32 = 1 << 2; + pub const first_descriptor: u32 = 1 << 3; + pub const second_address_chained: u32 = 1 << 4; + pub const end_of_ring: u32 = 1 << 5; + pub const card_error_summary: u32 = 1 << 30; + pub const owned_by_idmac: u32 = 1 << 31; + + /// `sdmmc_struct.h:43` SDMMC_DMA_MAX_BUF_LEN. `buffer1_size` is 13 bits wide, so 8191 would + /// fit; ESP-IDF splits at 4096 and so does the bound below. + pub const max_buffer_len: u32 = 4096; +}; + +/// Bytes of L2MEM this driver owns, and the whole of its dynamic memory: there is no allocator +/// here and no allocation anywhere in the transfer path. +/// +/// 2 KiB of payload is chosen against what sits above: ESP-Hosted's SDIO transport moves at most +/// one 1600-byte frame plus its 12-byte header per CMD53, and the largest single command this +/// driver can express in block mode is 4 blocks of 512. Anything larger is split across commands +/// by `transferChunked`, which is correct for both addressing modes, so the number is a +/// speed/footprint trade and not a limit. +pub const bounce_len: u32 = 2048; + +/// Descriptor and bounce buffer in one cache-line-aligned region, so the one-time maintenance call +/// is one call over one range whose base and length are both multiples of 64. +const DmaRegion = extern struct { + desc: Descriptor, + _pad: [cache_line - @sizeOf(Descriptor)]u8, + buf: [bounce_len]u8, +}; + +comptime { + std.debug.assert(@sizeOf(Descriptor) == 16); + std.debug.assert(@sizeOf(DmaRegion) % cache_line == 0); + // One descriptor is enough only while the bounce buffer fits in one. If `bounce_len` ever + // grows past 4096 this has to become a ring, and this line is what will say so. + std.debug.assert(bounce_len <= Descriptor.max_buffer_len); +} + +/// 2112 bytes: 16 of descriptor, 48 of padding to a cache line, 2048 of payload. +var dma: DmaRegion align(cache_line) = std.mem.zeroes(DmaRegion); + +/// Addresses are `usize` rather than `u32` all the way to the register write. On this target the +/// two are the same type; on the host, where the arithmetic in these helpers is unit-tested, +/// `@intCast` of a real 64-bit address would panic before the test could check anything. +inline fn cachedAddr(p: *const anyopaque) usize { + return @intFromPtr(p); +} + +/// The address the *CPU* must use for anything in the DMA region. The hardware gets the cached +/// address - that is not an inconsistency, it is what ESP-IDF does: `gdma_link.c:268-273` hands +/// `list->items` to the peripheral and `:159` writes through `list->items_nc`. The alias exists to +/// change how the CPU's loads and stores are treated, and a bus master is not the CPU. +inline fn uncachedAddr(p: *const anyopaque) usize { + return cachedAddr(p) +% @as(usize, non_cacheable_offset); +} + +/// An address as the 32-bit register field the hardware reads it through. +inline fn busAddr(p: *const anyopaque) u32 { + return @intCast(cachedAddr(p)); +} + +inline fn descNc() *volatile Descriptor { + return @ptrFromInt(uncachedAddr(&dma.desc)); +} + +inline fn bufNc() [*]volatile u8 { + return @ptrFromInt(uncachedAddr(&dma.buf)); +} + +/// Write back and invalidate the DMA region once, so that no dirty line from `_start`'s .bss clear +/// can later land on top of what the alias writes. After this, the cached alias of this region is +/// never touched again by anything in this file. +fn syncDmaRegionOnce() void { + const base = busAddr(&dma); + const len: u32 = @sizeOf(DmaRegion); + _ = romCacheWriteBackAddr(cache_map_l1_dcache, base, len); + _ = romCacheInvalidateAddr(cache_map_l1_dcache, base, len); +} + +// -------------------------------------------------------------------------------------- timing +// +// Every wait in this file is bounded, and bounded in time rather than in loop iterations: a spin +// count is a different number on every optimize level, and this board has no debugger, so a wait +// that never returns is indistinguishable from a crash. +// +// The timebase is the RISC-V `cycle` CSR, the unprivileged shadow of `mcycle`, which is what +// ESP-IDF itself reads on this part (`rv_utils.h`, because SOC_CPU_HAS_CSR_PC is not defined for +// the P4) and what `src/soc.zig:116-131` already uses. It is deliberately *not* `hal.systimer`: +// systimer's `init` pulses the peripheral's reset, which would make the timebase jump under any +// other user, and `systimer.read` returns null when nothing has brought it up - neither is a +// property a bus driver should impose on its caller. +// +// The CPU clock is whatever the bootloader left, measured at 90 MHz on this board and rated to +// 400. Deadlines are computed at the 400 MHz *ceiling*, so on real silicon every timeout below is +// between 1x and 4.4x longer than its nominal microseconds. That is the safe direction: a timeout +// that fires early would turn a slow card into a spurious failure, and a timeout 4x long still +// terminates. +const assumed_cpu_hz_max: u32 = 400_000_000; + +inline fn cycleLow() u32 { + return asm volatile ("csrr %[r], 0xC00" + : [r] "=r" (-> u32), + ); +} + +/// A bounded wait. 32 bits of cycle counter wrap after 10.7 s at the assumed ceiling, which is an +/// order of magnitude past the longest deadline here, and the wrapping subtraction is correct +/// across the wrap anyway. +const Deadline = struct { + start: u32, + budget: u32, + + inline fn init(us: u32) Deadline { + return .{ .start = cycleLow(), .budget = us *% (assumed_cpu_hz_max / 1_000_000) }; + } + + inline fn expired(self: Deadline) bool { + return (cycleLow() -% self.start) >= self.budget; + } +}; + +/// `sd_host_private.h:62` SD_HOST_SDMMC_RESET_TIMEOUT_US. +const reset_timeout_us: u32 = 5_000_000; +/// `sd_host_private.h:61` SD_HOST_SDMMC_START_CMD_TIMEOUT_US - how long the CIU may take to accept +/// a command word, which is a bus-side handshake and nothing to do with the card. +const start_cmd_timeout_us: u32 = 1_000_000; +/// How long to wait for the card's response after the command has been accepted. The controller +/// has its own response timeout (TMOUT.response_timeout, 255 card clocks) and raises RTO, so this +/// only has to cover the case where the controller itself never reports anything. +const command_done_timeout_us: u32 = 200_000; +/// Data phase. TMOUT.data_timeout is programmed to 100 ms of card clocks, matching +/// `sd_host_sdmmc.c:531-533`; this outer bound is twice that. +const data_done_timeout_us: u32 = 200_000; +/// How long the card may hold DAT0 low before a new data command. +const busy_timeout_us: u32 = 500_000; + +// ------------------------------------------------------------------------------------- geometry + +pub const Width = enum { one, four }; + +/// Slot 1's pads on this board, and the GPIO-matrix signal each carries. +/// +/// Slot 0 has a direct IO MUX function and slot 1 does not +/// (`sdmmc_ll.h:88` SDMMC_LL_SLOT_SUPPORT_GPIO_MATRIX(1) is 1, and `sdmmc_periph.c:37-49` has +/// -1 for every slot-1 IO MUX pin), so every slot-1 signal is routed through the matrix. The +/// indices are `gpio_sig_map.h:8-18`, reached here through `regs` rather than written out: the +/// same discipline `hal/gpio.zig` applies to SIG_GPIO_OUT_IDX, for the same reason. +pub const Pins = struct { + clk: u8, + cmd: u8, + d0: u8, + d1: u8, + d2: u8, + d3: u8, +}; + +/// The ESP32-C6 coprocessor's wiring on this board. CLK 18, CMD 19, D0-D3 = 14/15/16/17. +pub const c6_pins: Pins = .{ .clk = 18, .cmd = 19, .d0 = 14, .d1 = 15, .d2 = 16, .d3 = 17 }; + +const sig = struct { + const cclk: u32 = @intCast(regs.SD_CARD_CCLK_2_PAD_OUT_IDX); + const ccmd: u32 = @intCast(regs.SD_CARD_CCMD_2_PAD_OUT_IDX); + const cdata0: u32 = @intCast(regs.SD_CARD_CDATA0_2_PAD_OUT_IDX); + const cdata1: u32 = @intCast(regs.SD_CARD_CDATA1_2_PAD_OUT_IDX); + const cdata2: u32 = @intCast(regs.SD_CARD_CDATA2_2_PAD_OUT_IDX); + const cdata3: u32 = @intCast(regs.SD_CARD_CDATA3_2_PAD_OUT_IDX); + const card_detect: u32 = @intCast(regs.SD_CARD_DETECT_N_2_PAD_IN_IDX); + const card_int: u32 = @intCast(regs.SD_CARD_INT_N_2_PAD_IN_IDX); + + comptime { + // The `_2` in these names is slot 1: `sdmmc_periph.c:52-76` fills + // `sdmmc_slot_gpio_sig[1]` from exactly these macros. Slot 0's set is named `_1` and would + // route the wrong controller port to the C6's pads, silently. + std.debug.assert(cclk == 0 and ccmd == 1 and cdata0 == 2); + std.debug.assert(cdata1 == 3 and cdata2 == 4 and cdata3 == 5); + + // The card interrupt is sensed on D1's *input* index, and `configurePins` hands `matrixIn` + // the *output* one - correct only because the P4's two signal tables agree on this signal. + // `gpio_sig_map.h:13-14` gives cdata1 the number 3 in both directions, and ESP-IDF relies + // on the same coincidence: `configure_pin_gpio_matrix` (`sd_host_sdmmc.c:1091-1105`) passes + // one `gpio_matrix_sig` to both `esp_rom_gpio_connect_in_signal` and `..._out_signal`. + // Asserted rather than assumed, because a mismatch here would route data correctly and + // sense interrupts from the wrong pad - which is invisible until something waits. + std.debug.assert(cdata1 == @as(u32, @intCast(regs.SD_CARD_CDATA1_2_PAD_IN_IDX))); + } +}; + +// -------------------------------------------------------------------------------------- state + +const State = struct { + slot: u1 = 1, + width: Width = .four, + /// The frequency `cardInit` switches to once the card is addressed and in 4-bit mode. + target_khz: u32 = 40_000, + pins: Pins = c6_pins, + /// Relative card address from CMD3, needed as the argument of CMD7. + rca: u16 = 0, + initialised: bool = false, +}; + +var state: State = .{}; + +/// The card's relative address, as returned by CMD3. Zero until `cardInit` has run. +pub fn rca() u16 { + return state.rca; +} + +inline fn slotBit() u32 { + return @as(u32, 1) << state.slot; +} + +// ------------------------------------------------------------------------------ command words +// +// One word, one function, no hardware. This is the part of the driver most worth testing on the +// host, and the part the oracle can compare against ESP-IDF's own bitfield struct without going +// anywhere near the card. + +/// Compose one field's contribution to a register word. `mmio.Reg.write` does this against a +/// register; here the destination is a value, because the command word is built, checked and only +/// then stored. +inline fn bits(comptime f: Field, v: u32) u32 { + return (v & f.unshiftedMask()) << f.shift; +} + +const cmd_index = Field.of(regs.SDHOST_CMD_INDEX_S, regs.SDHOST_CMD_INDEX_V); +const response_expect = Field.of(regs.SDHOST_RESPONSE_EXPECT_S, regs.SDHOST_RESPONSE_EXPECT_V); +const response_length = Field.of(regs.SDHOST_RESPONSE_LENGTH_S, regs.SDHOST_RESPONSE_LENGTH_V); +const check_response_crc = Field.of(regs.SDHOST_CHECK_RESPONSE_CRC_S, regs.SDHOST_CHECK_RESPONSE_CRC_V); +const data_expected = Field.of(regs.SDHOST_DATA_EXPECTED_S, regs.SDHOST_DATA_EXPECTED_V); +const read_write = Field.of(regs.SDHOST_READ_WRITE_S, regs.SDHOST_READ_WRITE_V); +const transfer_mode = Field.of(regs.SDHOST_TRANSFER_MODE_S, regs.SDHOST_TRANSFER_MODE_V); +const send_auto_stop = Field.of(regs.SDHOST_SEND_AUTO_STOP_S, regs.SDHOST_SEND_AUTO_STOP_V); +const wait_prvdata_complete = Field.of(regs.SDHOST_WAIT_PRVDATA_COMPLETE_S, regs.SDHOST_WAIT_PRVDATA_COMPLETE_V); +const stop_abort_cmd = Field.of(regs.SDHOST_STOP_ABORT_CMD_S, regs.SDHOST_STOP_ABORT_CMD_V); +const send_initialization = Field.of(regs.SDHOST_SEND_INITIALIZATION_S, regs.SDHOST_SEND_INITIALIZATION_V); +const card_number = Field.of(regs.SDHOST_CARD_NUMBER_S, regs.SDHOST_CARD_NUMBER_V); +const update_clock_registers_only = Field.of(regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_S, regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_V); +/// `sdmmc_reg.h:486-494` spells this `USE_HOLE_REG`; `sdmmc_struct.h:473` spells it +/// `use_hold_reg`, which is what it is - the hold register that synchronises CMD and DATA to +/// cclk_out. Same bit 29, and ESP-IDF sets it on every command (`sd_host_sdmmc.c:859-860`). +const use_hold_reg = Field.of(regs.SDHOST_USE_HOLE_REG_S, regs.SDHOST_USE_HOLE_REG_V); +const start_cmd = Field.of(regs.SDHOST_START_CMD_S, regs.SDHOST_START_CMD_V); + +pub const Response = enum { none, short, long }; +pub const Direction = enum { read, write }; + +/// Everything that distinguishes one command from another, in the terms the register uses. +pub const Command = struct { + index: u6, + response: Response = .none, + /// Whether the controller checks the response's CRC7. Off for R3 and R4, which do not carry a + /// valid one - `sd_protocol_types.h:140-141` define both without SCF_RSP_CRC, and + /// `make_hw_cmd` (`sd_trans_sdmmc.c:214-216`) keys `check_response_crc` off exactly that flag. + check_crc: bool = false, + data: ?Direction = null, + /// 80 clocks of 1 before the command. Required once after power-on, and set only for CMD0, + /// which is where ESP-IDF sets it (`sd_trans_sdmmc.c:197-206`). + send_init: bool = false, + /// Wait for a previous data transfer to finish before sending. Set on everything except CMD0, + /// CMD12 and CMD11, again following `make_hw_cmd`. + wait_prvdata: bool = true, + auto_stop: bool = false, + stop_abort: bool = false, + /// Not a command at all: push CLKDIV/CLKSRC/CLKENA into the card clock domain. + update_clock: bool = false, + slot: u1 = 0, +}; + +/// The 32-bit word that, written to SDHOST_CMD_REG, issues `c`. +/// +/// This is `make_hw_cmd` (`sd_trans_sdmmc.c:190-229`) plus the three fields +/// `sd_host_slot_start_command` adds afterwards - `use_hold_reg`, `card_num` and `start_command` +/// (`sd_host_sdmmc.c:859-881`) - because those three are not optional and splitting them across +/// two functions is how one of them gets forgotten. +pub fn commandWord(c: Command) u32 { + var w: u32 = 0; + w |= bits(cmd_index, c.index); + if (c.response != .none) w |= bits(response_expect, 1); + if (c.response == .long) w |= bits(response_length, 1); + if (c.check_crc) w |= bits(check_response_crc, 1); + if (c.data) |dir| { + w |= bits(data_expected, 1); + if (dir == .write) w |= bits(read_write, 1); + } + if (c.auto_stop) w |= bits(send_auto_stop, 1); + if (c.wait_prvdata) w |= bits(wait_prvdata_complete, 1); + if (c.stop_abort) w |= bits(stop_abort_cmd, 1); + if (c.send_init) w |= bits(send_initialization, 1); + if (c.update_clock) w |= bits(update_clock_registers_only, 1); + w |= bits(card_number, c.slot); + // Block transfers only; `transfer_mode` selects stream mode, which no SDIO command uses. + w |= bits(transfer_mode, 0); + w |= bits(use_hold_reg, 1); + w |= bits(start_cmd, 1); + return w; +} + +// ------------------------------------------------------------------------------ SDIO protocol +// +// Command indices and argument layouts, from `sd_protocol_defs.h`. Written out as constants rather +// than reached through `regs` because they are the SD specification, not this chip: the register +// headers know nothing about them. + +/// `sd_protocol_defs.h:35`, `:40`, `:61`, `:78-80`. +const cmd_go_idle_state: u6 = 0; +const cmd_send_relative_addr: u6 = 3; +const cmd_io_send_op_cond: u6 = 5; +const cmd_select_card: u6 = 7; +const cmd_io_rw_direct: u6 = 52; +const cmd_io_rw_extended: u6 = 53; + +/// CMD52's argument: `sd_protocol_defs.h:484-492`. +pub fn cmd52Arg(write: bool, func: u3, addr: u17, raw_flag: bool, data: u8) u32 { + var a: u32 = 0; + if (write) a |= @as(u32, 1) << 31; + a |= @as(u32, func) << 28; + if (raw_flag) a |= @as(u32, 1) << 27; + a |= @as(u32, addr) << 9; + a |= data; + return a; +} + +/// CMD53's argument: `sd_protocol_defs.h:496-506`. +/// +/// `count` is blocks in block mode and bytes in byte mode, and it is 9 bits: 0 means 512 in byte +/// mode ("See 5.3.1 SDIO simplified spec", `sdmmc_io.c:351-355`) and infinite in block mode, which +/// this driver never asks for. +pub fn cmd53Arg(write: bool, func: u3, addr: u17, block_mode: bool, incrementing: bool, count: u9) u32 { + var a: u32 = 0; + if (write) a |= @as(u32, 1) << 31; + a |= @as(u32, func) << 28; + if (block_mode) a |= @as(u32, 1) << 27; + if (incrementing) a |= @as(u32, 1) << 26; + a |= @as(u32, addr) << 9; + a |= count; + return a; +} + +/// The block size this driver programmes into BLKSIZ and into the card's CCCR/FBR. +/// `sdmmc_common.h:195` SDMMC_IO_BLOCK_SIZE, and ESP-Hosted writes the same 512 into FN0 and FN1 +/// (`port_esp_hosted_host_sdio.c:211-217`). +pub const io_block_size: u32 = 512; + +/// CCCR register offsets, `sd_protocol_defs.h:509-530`. +pub const cccr = struct { + pub const revision: u17 = 0x00; + pub const fn_enable: u17 = 0x02; + pub const fn_ready: u17 = 0x03; + pub const int_enable: u17 = 0x04; + pub const int_pending: u17 = 0x05; + pub const ctl: u17 = 0x06; + pub const bus_width: u17 = 0x07; + pub const card_cap: u17 = 0x08; + pub const cis_ptr: u17 = 0x09; + pub const blksize_l: u17 = 0x10; + pub const blksize_h: u17 = 0x11; + + pub const ctl_reset: u8 = 1 << 3; + pub const bus_width_1: u8 = 0; + pub const bus_width_4: u8 = 2; + /// Low-speed card; and "4-bit low speed", which says a low-speed card supports 4 bits anyway. + pub const card_cap_lsc: u8 = 1 << 6; + pub const card_cap_4bls: u8 = 1 << 7; +}; + +/// `sd_protocol_defs.h:533` SD_IO_FBR_START - function n's register block starts here. +const fbr_start: u17 = 0x100; + +/// R4's fields, `sd_protocol_defs.h:478-481`. +const r4_mem_ready: u32 = 1 << 31; +const r4_mem_present: u32 = 1 << 27; + +/// The voltage window the host offers in CMD5's second pass: bits 23:15, i.e. 2.8-3.6 V. +/// `sd_protocol_defs.h:109` SD_OCR_VOL_MASK, which is the whole of what `get_host_ocr` returns - +/// "For now tell that the host has 2.8-3.6V voltage range" (`sdmmc_common.h:174-180`). +const host_ocr: u32 = 0x00ff_8000; + +// ------------------------------------------------------------------------------- command issue + +/// Write one command word and wait for the CIU to take it. No card traffic is implied: a clock +/// update command goes through here too. +/// +/// Both waits are the ones `sd_host_slot_start_command` performs (`sd_host_sdmmc.c:862-892`), +/// bounded the same way. The first is not redundant with the second: writing any command register +/// while `start_command` is still set is a hardware locked write error, and HLE is reported +/// asynchronously in RINTSTS where it is easy to attribute to the wrong command. +fn startCommand(word: u32, arg: u32) Error!void { + var d = Deadline.init(start_cmd_timeout_us); + while (cmd.get(start_cmd) != 0) { + if (d.expired()) return error.Busy; + } + cmdarg.writeRaw(arg); + cmd.writeRaw(word); + d = Deadline.init(start_cmd_timeout_us); + while (cmd.get(start_cmd) != 0) { + if (d.expired()) return error.Timeout; + } +} + +/// Push CLKDIV, CLKSRC and CLKENA into the card clock domain. +fn clockUpdate() Error!void { + try startCommand(commandWord(.{ + .index = 0, + .update_clock = true, + .wait_prvdata = true, + .slot = state.slot, + }), 0); +} + +/// Turn a RINTSTS snapshot into the failure it describes. +/// +/// Order matters only in that the first match wins, and it is chosen so the most specific cause is +/// reported: a CRC error and a timeout together is a CRC error, because the timeout is downstream +/// of it. +fn decodeErrors(sts: u32) Error!void { + if (sts & (Event.rcrc | Event.dcrc) != 0) return error.CrcError; + if (sts & (Event.rto | Event.drto | Event.hto) != 0) return error.Timeout; + if (sts & (Event.re | Event.hle | Event.ebe | Event.sbe | Event.frun) != 0) return error.ResponseError; +} + +/// Wait for one or more RINTSTS bits, failing on any error bit or on the deadline. +/// +/// RINTSTS is write-1-to-clear, so this reads with `raw()` and clears with `writeRaw(mask)` - +/// never `modify`, which would clear every bit it read back and lose the events this function is +/// not waiting for. +fn waitEvents(want: u32, errors: u32, us: u32) Error!u32 { + const d = Deadline.init(us); + while (true) { + const sts = rintsts.raw(); + if (sts & errors != 0) { + rintsts.writeRaw(sts & (want | errors)); + try decodeErrors(sts & errors); + // Every bit any caller passes in `errors` is covered above; a new one arriving here + // is a bug in this file, and reporting it beats an `unreachable` on a board with no + // debugger. + return error.ResponseError; + } + if (sts & want == want) { + rintsts.writeRaw(want); + return sts; + } + if (d.expired()) return error.Timeout; + } +} + +/// A command with no data phase: issue it, wait for command-done, return R1/R5's first word. +fn sendCommand(c: Command, arg: u32) Error!u32 { + // Everything this command is about to overwrite. This slot's SDIO card interrupt is + // deliberately left alone - the C6 raises it asynchronously and clearing it here would drop a + // wakeup the layer above is waiting for - and `clearNonSlaveInterrupts` is exactly that set. + // + // It used to be `Event.default & ~Event.cd`, which is a *subset* of the event bits and left + // four of them latched for ever: txdr(4), rxdr(5), frun(11) and acd(14). Two consequences, one + // cosmetic and one not. Cosmetic: every RINTSTS a diagnostic prints carries a stale 0x10 from + // the first transfer onwards, which is noise in exactly the register that has to be read + // carefully. Not cosmetic: **frun is a member of `Event.data_errors`**, so one FIFO + // under/overrun - ever - would latch a bit that nothing clears and fail every subsequent + // `waitEvents(Event.dto, Event.data_errors, ...)` for the rest of the run. The data path works + // today only because frun has never fired. + clearNonSlaveInterrupts(); + var cc = c; + cc.slot = state.slot; + try startCommand(commandWord(cc), arg); + _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us); + return resp0.raw(); +} + +/// R5's status byte, the one CMD52 and CMD53 return. `sd_protocol_defs.h:493` takes the data byte; +/// the flags above it say whether the card accepted the command at all. +const r5_com_crc_error: u32 = 1 << 15; +const r5_illegal_command: u32 = 1 << 14; +const r5_error: u32 = 1 << 11; +const r5_function_number: u32 = 1 << 9; +const r5_out_of_range: u32 = 1 << 8; +const r5_bad: u32 = r5_com_crc_error | r5_illegal_command | r5_error | r5_function_number | r5_out_of_range; + +fn checkR5(r: u32) Error!u8 { + if (r & r5_com_crc_error != 0) return error.CrcError; + if (r & r5_bad != 0) return error.ResponseError; + return @truncate(r); +} + +// ------------------------------------------------------------------------------- bring-up + +/// Controller, FIFO and DMA reset, then wait for all three to self-clear. +/// +/// All three bits are self-clearing, and `sdmmc_ll.h:486`, `:510` and `:534` each say so with a +/// different delay ("two AHB clock cycles", "after reset done"). ESP-IDF sets all three and polls +/// all three together (`sd_host_sdmmc.c:917-950`), which is what makes one bounded wait correct +/// for the set. +pub fn resetController() Error!void { + ctrl.modify(.{ controller_reset.is(1), fifo_reset.is(1), dma_reset.is(1) }); + const d = Deadline.init(reset_timeout_us); + while (true) { + const v = ctrl.raw(); + if (v & (controller_reset.mask() | fifo_reset.mask() | dma_reset.mask()) == 0) return; + if (d.expired()) return error.Timeout; + } +} + +/// The interrupt configuration `sd_host_sdmmc.c:120-124` establishes - clear everything, mask +/// everything, then unmask the completion and error events and turn the global enable on - with +/// one deliberate deviation: card detect stays masked *and* gets cleared. See `Event.armed` for +/// why that bit is load-bearing on a board with no card-detect pin. +/// +/// `int_enable` gates the controller's single line into the CLIC. It is on even though this driver +/// polls, because RINTSTS is set regardless and the layer above may register a handler for the +/// SDIO card interrupt; leaving it off would mean `setSlaveInterruptEnabled(true)` silently did +/// nothing. +pub fn configureInterrupts() void { + rintsts.writeRaw(0xffff_ffff); + intmask.writeRaw(0); + ctrl.modify(.{int_enable.is(0)}); + intmask.writeRaw(Event.armed); + // Belt and braces: `armed` keeps the controller from reporting a latched cd, and this makes + // sure there is no latched cd to report if anything ever unmasks it again. + rintsts.writeRaw(Event.cd); + ctrl.modify(.{int_enable.is(1)}); +} + +/// `sdmmc_ll_init_dma`, `sdmmc_ll.h:796-804`: enable the DMA path, clear the bus-mode register, +/// pulse the IDMAC's own software reset, and unmask its three completion interrupts. +pub fn initDma() void { + ctrl.modify(.{dma_enable.is(1)}); + bmod.writeRaw(0); + bmod.modify(.{bmod_swr.is(1)}); + idinten.modify(.{ idinten_ni.is(1), idinten_ri.is(1), idinten_ti.is(1) }); +} + +/// Leave the controller's interrupt output silent, and both status registers clean. +/// +/// `configureInterrupts` and `initDma` above are ESP-IDF's sequences, and ESP-IDF is +/// interrupt-driven: its transfers wait on a queue its ISR fills, so it needs command-done, the +/// error bits and the IDMAC's completions in the masks. **This driver polls**, so every one of +/// those is noise on a line whose only handler understands one cause. Worse than noise: two of +/// them hold the line asserted forever. +/// +/// * **INTMASK** gates RINTSTS into MINTSTS. Zero here costs nothing - `waitEvents` reads +/// RINTSTS, and "Bits are logged regardless of interrupt mask status" +/// (`sdmmc_struct.h:589-591`). +/// * **IDINTEN** gates the IDMAC's own events, and it does *not* go through INTMASK. `initDma` +/// enables NI/RI/TI because `sdmmc_ll_init_dma` does, and IDF can afford that because its ISR +/// clears IDSTS on every interrupt (`sd_host_sdmmc.c:801-802`). `dataTransfer` clears IDSTS +/// *before* a transfer and nothing clears it after, so RI and its sticky summary NIS stay set +/// from the first CMD53 onwards - a permanently asserted interrupt line that no INTMASK write +/// can lower. +/// +/// `CTRL.int_enable` stays on: with both masks at zero the line cannot assert anyway, and leaving +/// the global enable alone keeps `armSlaveInterrupt` down to the stores that matter. +pub fn muteInterrupts() void { + intmask.writeRaw(0); + idinten.writeRaw(0); + rintsts.writeRaw(0xffff_ffff); + idsts.writeRaw(idsts_event_mask); +} + +/// FIFO watermarks and DMA burst size. +/// +/// ESP-IDF never writes this register on any target - there is no `sdmmc_ll` function for it and +/// no assignment anywhere in `components/` - so the value in the measured working dump, +/// `fifoth=0x01FF0000`, is the hardware's reset state: rx watermark 511, tx watermark 0, burst +/// size code 0 (one transfer). This function writes that value explicitly rather than inheriting +/// it, because a controller reset is not the only thing that can have touched the register and +/// "the same as reset" is a claim worth making in code. +/// +/// It is also a performance knob left deliberately untouched: DesignWare recommends half the FIFO +/// depth for both watermarks and a burst size matching the AXI port, and tx watermark 0 means a +/// DMA request only when the FIFO is completely empty. Turning that knob without a board to +/// measure on would be guessing, and the guess would be against a configuration known to work at +/// 40 MHz. +pub fn setFifoThreshold(rx: u32, tx: u32, msize: u32) void { + fifoth.write(.{ rx_wmark.is(rx), tx_wmark.is(tx), dma_msize.is(msize) }); +} + +/// The reset-value watermarks, which are the ones the working dump shows. +pub const default_rx_watermark: u32 = 511; +pub const default_tx_watermark: u32 = 0; +pub const default_dma_msize: u32 = 0; + +/// Bus width, host side. The card side is a CCCR write and is done in `cardInit`; the two must +/// change in that order, or the next command goes out on a bus the card is not listening to. +pub fn setBusWidth(w: Width) void { + const m = slotBit(); + const c8 = ctype.get(card_width8) & ~m; + const c4 = switch (w) { + .one => ctype.get(card_width4) & ~m, + .four => ctype.get(card_width4) | m, + }; + ctype.modify(.{ card_width4.is(c4), card_width8.is(c8) }); +} + +pub fn setBlockSize(bytes: u32) void { + blksiz.modify(.{block_size.is(bytes)}); +} + +/// The two-stage divider, resolved. Stage one is `host_div` in HP_SYS_CLKRST, stage two is the +/// controller's own CLKDIV, and the card clock is `160 MHz / host_div / (2 * card_div)` with +/// `card_div == 0` meaning bypass. +/// +/// The table is `sd_host_slot_get_clk_dividers` (`sd_host_sdmmc.c:998-1062`), restricted to the +/// PLL160M source: this board's C6 is a 3.3 V SDIO device, so the 200 MHz SDIO PLL and the UHS-I +/// speeds it exists for are out of reach and out of scope. +pub const Dividers = struct { host: u32, card: u32 }; + +pub fn dividersFor(khz: u32) Dividers { + const src_hz: u32 = 160_000_000; + if (khz >= 40_000) return .{ .host = 4, .card = 0 }; // 160/4 = 40 MHz + if (khz == 20_000) return .{ .host = 8, .card = 0 }; // 160/8 = 20 MHz + if (khz == 400) return .{ .host = 10, .card = 20 }; // 160/10/(20*2) = 400 kHz + var host = src_hz / (khz * 1000); + var card: u32 = 0; + if (host > 15) { + host = 2; + card = (src_hz / 2) / (2 * khz * 1000); + if (((src_hz / 2) % (2 * khz * 1000)) > 0) card += 1; + } else if (src_hz % (khz * 1000) > 0) { + host += 1; + } + return .{ .host = host, .card = card }; +} + +/// Stage one: the clock generator in HP_SYS_CLKRST. `sdmmc_ll_set_clock_div`, +/// `sdmmc_ll.h:244-258`. +/// +/// The `edge_cfg_update` bit is write-to-trigger and must be pulsed - set then cleared - after the +/// three edge fields, or the new division is programmed and never latched. +pub fn setHostClockDiv(div: u32) void { + if (div > 1) { + peri_clk_ctrl02.modify(.{ + sdio_ls_clk_edge_h.is(div / 2 - 1), + sdio_ls_clk_edge_n.is(div - 1), + sdio_ls_clk_edge_l.is(div - 1), + }); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)}); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)}); + } else { + peri_clk_ctrl01.modify(.{sdio_hs_mode.is(1)}); + peri_clk_ctrl02.modify(.{ + sdio_ls_clk_edge_h.is(0), + sdio_ls_clk_edge_n.is(0), + sdio_ls_clk_edge_l.is(0), + }); + } +} + +/// PLL160M, the only source this driver uses. `sdmmc_ll_select_clk_source`, `sdmmc_ll.h:212-229`: +/// source value 0 is PLL160M and 1 is the 200 MHz SDIO PLL. +pub fn selectPll160m() void { + peri_clk_ctrl01.modify(.{ sdio_ls_clk_src_sel.is(0), sdio_ls_clk_en.is(1) }); +} + +/// The driving, sampling and self clocks the pad logic runs on. `sdmmc_ll_init_phase_delay`, +/// `sdmmc_ll.h:303-315`. Without this the three gates stay off and the bus does not move, which is +/// the kind of failure that looks like a wiring fault. +pub fn initPhaseDelay() void { + peri_clk_ctrl02.modify(.{ + sdio_ls_drv_clk_en.is(1), + sdio_ls_sam_clk_en.is(1), + sdio_ls_slf_clk_en.is(1), + sdio_ls_drv_clk_edge_sel.is(1), + sdio_ls_sam_clk_edge_sel.is(0), + sdio_ls_slf_clk_edge_sel.is(0), + }); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)}); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)}); +} + +/// Stage one, whole: divider, source, phase clocks, and the settle the hardware needs afterwards. +/// `sd_host_set_clk_div`, `sd_host_sdmmc.c:974-990`, including its closing +/// `esp_rom_delay_us(10)` - "Wait for the clock to propagate". +/// +/// This has to happen before the controller reset, not after. `controller_reset` is documented to +/// self-clear "after two AHB and two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`), so with +/// no card clock reaching the block the bit never clears and the reset wait runs to its full +/// timeout. ESP-IDF's order says the same thing without saying it: `sd_host_set_clk_div` at +/// `sd_host_sdmmc.c:109`, `sd_host_reset` at `:112`. +pub fn setHostClock(div: u32) void { + setHostClockDiv(div); + selectPll160m(); + initPhaseDelay(); + spinMicros(10); +} + +/// Stage two: the controller's per-slot divider and the divider-to-slot mux. +/// `sdmmc_ll_set_card_clock_div`, `sdmmc_ll.h:431-442`. Slot 1 uses divider 1, slot 0 uses divider +/// 0 - so the mux value equals the slot number, which is why one line covers both. +pub fn setCardClockDiv(div: u32) void { + if (state.slot == 0) { + clksrc.modify(.{clksrc_card0.is(0)}); + clkdiv.modify(.{clk_divider0.is(div)}); + } else { + clksrc.modify(.{clksrc_card1.is(1)}); + clkdiv.modify(.{clk_divider1.is(div)}); + } +} + +/// The card clock's on/off switch, one bit per slot. Takes effect only after a clock update +/// command. `sdmmc_ll_enable_card_clock`, `sdmmc_ll.h:415-422`. +pub fn setCardClockEnabled(on: bool) void { + const cur = clkena.get(cclk_enable); + clkena.modify(.{cclk_enable.is(if (on) cur | slotBit() else cur & ~slotBit())}); +} + +/// Stop the card clock while the card is idle. `sdmmc_ll_enable_card_clock_low_power`, +/// `sdmmc_ll.h:474-481`. **Off** for SDIO: the card raises its interrupt on D1 and cannot do so +/// with the clock stopped, which is why ESP-IDF clears the same bit for any slot with +/// `cclk_always_on` (`sd_host_sdmmc.c:272-285`) and why the measured working dump reads +/// `clkena=0x00000002` rather than `0x00020002`. +pub fn setCardClockLowPower(on: bool) void { + const cur = clkena.get(lp_enable); + clkena.modify(.{lp_enable.is(if (on) cur | slotBit() else cur & ~slotBit())}); +} + +/// Bytes in the next data transfer. `sdmmc_ll_set_data_transfer_len`, `sdmmc_ll.h:651-654`. +pub fn setDataTransferLen(len: u32) void { + bytcnt.modify(.{byte_count.is(len)}); +} + +/// Data-read and response timeouts, both in card output clocks. +/// `sdmmc_ll_set_data_timeout` / `sdmmc_ll_set_response_timeout`, `sdmmc_ll.h:564-582`. +pub fn setTimeouts(data_cycles: u32, response_cycles: u32) void { + tmout.write(.{ + data_timeout.is(if (data_cycles > 0xff_ffff) 0xff_ffff else data_cycles), + response_timeout.is(response_cycles), + }); +} + +/// Turn the internal DMA path on or off: both CTRL bits and both BMOD bits, together. +/// `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`. +pub fn setDmaEnabled(on: bool) void { + const v: u32 = @intFromBool(on); + ctrl.modify(.{ dma_enable.is(v), use_internal_dma.is(v) }); + bmod.modify(.{ bmod_de.is(v), bmod_fb.is(v) }); +} + +/// Where the IDMAC fetches its first descriptor. `sdmmc_ll_set_desc_addr`, `sdmmc_ll.h:673-676`. +/// The address is the *cached* one; see the "Cache" section above for why that is right. +pub fn setDescriptorAddr(a: u32) void { + dbaddr.writeRaw(a); +} + +/// Change the card clock, safely: stop it, reprogramme both stages, start it again, with a clock +/// update command after each step. `sd_host_slot_set_card_clk`, `sd_host_sdmmc.c:487-537`. +/// +/// Low-power mode is left **off**, which is the one place this deviates from a plain SD host and +/// matches the measured dump (`clkena=0x00000002`: clock enabled for slot 1, `lp_enable` clear). +/// `clkena.lp_enable` stops cclk while the card is idle; an SDIO card signals its interrupt on D1 +/// and needs the clock running to do it, which is why ESP-IDF turns the same bit off for any slot +/// with `cclk_always_on` (`sd_host_sdmmc.c:272-285`). +pub fn setBusClock(khz: u32) Error!void { + const d = dividersFor(khz); + + setCardClockEnabled(false); + try clockUpdate(); + + setCardClockDiv(d.card); + setHostClock(d.host); + try clockUpdate(); + + setCardClockEnabled(true); + setCardClockLowPower(false); + try clockUpdate(); + + // 100 ms of card clocks for data, and the maximum 255 card clocks for a response - "always set + // response timeout to highest value, it's small enough anyway" (`sd_host_sdmmc.c:534-535`). + setTimeouts(100 * khz, 255); +} + +/// Route slot 1's six signals to the C6's pads. +/// +/// Pull-ups: **the board provides them externally and this enables the internal ones anyway**, on +/// all six pads, because that is what the working configuration does. It is not obvious from +/// ESP-Hosted's side - it leaves `SDMMC_SLOT_FLAG_INTERNAL_PULLUP` clear +/// (`SDMMC_SLOT_CONFIG_DEFAULT`, `sdmmc_default_configs.h:98`: `.flags = 0`) - but every pad still +/// gets one, because `configure_pin_gpio_matrix` opens with `gpio_reset_pin` +/// (`sd_host_sdmmc.c:1096`) and that function enables the pull-up unconditionally: "for powersave +/// reasons, the GPIO should not be floating, select pullup" (`gpio.c:469-472`). The 40 MHz link +/// that produced the register dump therefore had both the module's external pull-ups and these. +/// Matching a measured configuration beats reasoning about which resistor is redundant. +/// +/// D1 has a second job: it is the SDIO interrupt line, and the controller derives that interrupt +/// from the same routed data signal - `sd_host_slot_sdmmc_io_int_enable` (`sd_host_sdmmc.c:381-388`) +/// is *only* `configure_pin(d1, sdmmc_slot_gpio_sig[slot].d1, GPIO_MODE_INPUT_OUTPUT)`, the same +/// two matrix writes and the same `fun_ie` the loop below already does, and it touches no +/// controller register at all. Both halves are load-bearing and neither is visible in a working +/// data path: with `fun_ie` clear, or with the *input* side of the matrix left pointing elsewhere, +/// D1 still drives and every transfer still completes while the controller samples a constant and +/// never latches a card interrupt. A link that carries traffic and never reports an event is +/// exactly what that failure looks like, which is why D1 is routed both ways even in 1-bit mode +/// and why `interruptDiagnostics` prints both bits. +/// +/// D3 is *not* routed to the controller yet. It is driven high as a plain GPIO output until the +/// bus is switched to 4 bits, which is how a host tells an SDIO card to use SD mode rather than +/// SPI mode; `sd_host_sdmmc.c:1282-1294` does the same and `cardInit` reconnects it at +/// `sd_host_sdmmc.c:575-583`'s point in the sequence. +pub fn configurePins(pins: Pins) void { + // CLK is output-only. + gpio.matrixOut(pins.clk, sig.cclk); + gpio.setInputEnable(pins.clk, false); + gpio.setPull(pins.clk, .up); + + const bidir = [_]struct { pin: u8, signal: u32 }{ + .{ .pin = pins.cmd, .signal = sig.ccmd }, + .{ .pin = pins.d0, .signal = sig.cdata0 }, + .{ .pin = pins.d1, .signal = sig.cdata1 }, + .{ .pin = pins.d2, .signal = sig.cdata2 }, + }; + for (bidir) |b| { + gpio.matrixOut(b.pin, b.signal); + gpio.matrixIn(b.pin, b.signal); + gpio.setInputEnable(b.pin, true); + gpio.setPull(b.pin, .up); + } + + // D3 high, as a GPIO, until the bus width changes. + gpio.configureOutput(pins.d3, .{ .readback = true }); + gpio.setPull(pins.d3, .up); + gpio.setHigh(pins.d3); + + // Card detect and the card's own interrupt-request pin are not wired to anything on this + // board, so both are tied off in the matrix exactly as ESP-IDF ties them when no pin is + // configured: card-detect to a constant 0 ("card present", `sd_host_sdmmc.c:1315-1319`) and + // card-int-n to a constant 1, i.e. inactive (`:1304-1306`). Leaving them unrouted is not the + // same thing: GPIO_FUNCn_IN_SEL_CFG resets with `sig_in_sel` clear, which bypasses the matrix + // and takes the signal from whatever direct pad function exists - and slot 1 has none. + // Write protect is left alone; nothing in this driver reads WRTPRT. + gpio.matrixIn(gpio.matrix_const_zero, sig.card_detect); + gpio.matrixIn(gpio.matrix_const_one, sig.card_int); +} + +/// Reconnect D3 to the controller, once the card is in 4-bit mode. +fn attachD3() void { + gpio.matrixOut(state.pins.d3, sig.cdata3); + gpio.matrixIn(state.pins.d3, sig.cdata3); + gpio.setInputEnable(state.pins.d3, true); + gpio.setPull(state.pins.d3, .up); +} + +/// Everything from the clock gate to a controller that will accept a command, with the bus at the +/// 400 kHz probing frequency and 1 bit wide - which is where an SDIO card has to be met. +/// +/// `cardInit` is what raises it to `khz` and to `width`, after the card has been addressed. +pub fn init(opts: struct { + slot: u1 = 1, + width: Width = .four, + khz: u32 = 40_000, + pins: Pins = c6_pins, +}) Error!void { + state = .{ + .slot = opts.slot, + .width = opts.width, + .target_khz = opts.khz, + .pins = opts.pins, + }; + + // The C6 hangs off slot 1 and slot 0's pads are the P4's own flash on most boards; refusing + // here is cheaper than debugging a bricked boot. + if (opts.slot != 1) return error.NotSupported; + + // 1. Bus clock and reset. Unlike most of this chip, SDMMC's bus clock is gated *off* at + // power-on (HP_SYS_CLKRST SOC_CLK_CTRL1 REG_SDMMC_SYS_CLK_EN, default 0), so this is a + // prerequisite and not a formality - without it the register block reads stale nonsense. + // Its reset bit is not in HP_SYS_CLKRST at all but in LP_AON_CLKRST; see hal/clkrst.zig. + clkrst.init(.sdmmc); + + // 2. The host clock generator, *before* the controller reset and not after. `sd_host_reset` + // polls three self-clearing bits, and `controller_reset` clears only "after two AHB and + // two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`) - with no card clock reaching the + // block that poll runs to its full timeout. ESP-IDF's controller init has the same order: + // `sd_host_set_clk_div(ctlr, SDMMC_CLK_SRC_DEFAULT, 2)` at `sd_host_sdmmc.c:109`, then + // `sd_host_reset` at `:112`. Divider 2 is IDF's provisional value, replaced at step 6. + setHostClock(2); + + // 3. Controller, FIFO and DMA out of reset. + try resetController(); + + // 4. Interrupts and DMA, before any command can produce one. The first two reproduce ESP-IDF + // for the differential; `muteInterrupts` then takes back everything this driver polls for + // instead of being interrupted by, leaving the line into the CLIC silent until a waiter + // arms it. + configureInterrupts(); + initDma(); + muteInterrupts(); + + // 5. Pads. After the clock gate so the controller's outputs are real, before the card clock so + // the first cycle the C6 sees is a clean one. + configurePins(state.pins); + + // 6. Bus clock at probing speed, 1 bit wide. An SDIO card has to be met at 400 kHz in 1-bit + // mode; `cardInit` raises both once the card has been addressed. + try setBusClock(400); + setBusWidth(.one); + + // 7. Transfer geometry. + setBlockSize(io_block_size); + setFifoThreshold(default_rx_watermark, default_tx_watermark, default_dma_msize); + setDescriptorAddr(busAddr(&dma.desc)); + + // 8. The one cache operation in this driver's life. See the "Cache" section above. + syncDmaRegionOnce(); + + state.initialised = true; + + // The one claim in this sequence with no differential case behind it, said out loud once, at + // the moment it is true. `configureInterrupts` and `initDma` are compared against ESP-IDF on + // the die; `muteInterrupts` cannot be - it is a deliberate deviation from IDF's ISR-driven + // design, and a reference implementation of our own decision would prove nothing. This line is + // the substitute, and it is worth a print because both zeros are load-bearing: a non-zero + // idinten here is an interrupt line that no INTMASK write can ever lower. + note("MARK SDMMC_INIT intmask=0x%08x idinten=0x%08x expect 0x00000000 and 0x00000000\r\n", .{ + intmask.raw(), idinten.raw(), + }); +} + +// ------------------------------------------------------------------------------ card bring-up + +/// CMD0, CMD5, CMD3, CMD7, then the CCCR writes that make function 1 usable: the sequence that +/// takes the C6 from "powered" to "answers CMD52". +/// +/// The command half follows `sdmmc_card_init` (`sdmmc_init.c:78-133`) restricted to the SDIO path: +/// `sdmmc_io_reset`, CMD0, `sdmmc_init_io` (CMD5 twice), `sdmmc_init_rca` (CMD3), +/// `sdmmc_init_select_card` (CMD7). The CCCR half is ESP-Hosted's `hosted_sdio_card_fn_init` +/// (`port_esp_hosted_host_sdio.c:143-220`) - enable function 1, wait for it to report ready, +/// unmask its interrupt, switch to 4 bits, set both block sizes to 512 - because that is what this +/// particular device needs and IDF's generic SDIO init does not do. +/// +/// The CMD52 that resets the card is allowed to fail. A device that is already out of reset +/// answers it; one that is not may time out, and `sdmmc_io_reset` (`sdmmc_io.c:66-83`) accepts +/// exactly that. +pub fn cardInit() Error!void { + if (!state.initialised) return error.NotSupported; + + // CCCR CTL bit 3: I/O reset. Best-effort, as above. + cmd52Write(0, cccr.ctl, cccr.ctl_reset) catch {}; + + // CMD0 with the 80-clock init sequence and no response. + _ = try sendCommand(.{ + .index = cmd_go_idle_state, + .response = .none, + .send_init = true, + .wait_prvdata = false, + }, 0); + // SDMMC_GO_IDLE_DELAY_MS (`sdmmc_common.h:34`), which `sdmmc_send_cmd_go_idle_state` waits + // out before returning (`sdmmc_cmd.c:114-116`). CMD0 has no response, so there is nothing to + // wait *for*: this is the card's own settling time and skipping it makes the next command a + // coin toss. + spinMicros(20_000); + + // CMD5 with a zero argument asks "are you an IO card, and what voltages do you take"; R4 has + // no CRC, hence `check_crc = false` (`sd_protocol_types.h:141`). + const probe = try sendCommand(.{ + .index = cmd_io_send_op_cond, + .response = .short, + .check_crc = false, + }, 0); + const functions = (probe >> 28) & 0x7; + if (functions == 0) return error.NotSupported; // answered CMD5, but has no IO function + + // CMD5 again with the voltage window, until the card reports ready. 100 attempts is + // `sdmmc_io.c:240`; the 10 ms between them is SDMMC_IO_SEND_OP_COND_DELAY_MS + // (`sdmmc_common.h:35`), spent here as a bounded spin rather than a scheduler delay. + const ocr = host_ocr & probe; + var ready = false; + var tries: u32 = 0; + while (tries < 100) : (tries += 1) { + const r = try sendCommand(.{ + .index = cmd_io_send_op_cond, + .response = .short, + .check_crc = false, + }, ocr); + if (r & r4_mem_ready != 0) { + ready = true; + break; + } + spinMicros(10_000); + } + if (!ready) return error.Timeout; + + // CMD3: the card picks its own relative address and returns it in R6[31:16]. + const r6 = try sendCommand(.{ + .index = cmd_send_relative_addr, + .response = .short, + .check_crc = true, + }, 0); + state.rca = @truncate(r6 >> 16); + + // CMD7 with that address moves the card from stand-by to transfer state. Every CMD52 and + // CMD53 after this is addressed to it implicitly. + _ = try sendCommand(.{ + .index = cmd_select_card, + .response = .short, + .check_crc = true, + }, @as(u32, state.rca) << 16); + + // ---- CCCR: function 1 on. + const ioe = try cmd52Read(0, cccr.fn_enable); + try cmd52Write(0, cccr.fn_enable, ioe | 0x02); + + // Wait for IOR bit 1. ESP-Hosted polls with a 10 ms gap and gives up after SDIO_INIT_MAX_RETRY + // (`port_esp_hosted_host_sdio.c:177-192`). + var fn_ready = false; + tries = 0; + while (tries < 100) : (tries += 1) { + if ((try cmd52Read(0, cccr.fn_ready)) & 0x02 != 0) { + fn_ready = true; + break; + } + spinMicros(10_000); + } + if (!fn_ready) return error.Timeout; + + // Master interrupt enable plus function 1's, so the C6 can raise D1. + const ie = try cmd52Read(0, cccr.int_enable); + try cmd52Write(0, cccr.int_enable, ie | 0x01 | 0x02); + + // ---- Bus width: card first, then host, then D3 joins the bus. + if (state.width == .four) { + const cap = try cmd52Read(0, cccr.card_cap); + // "Not a low-speed card" or "a low-speed card that supports 4 bits" - `sdmmc_io.c:182-183`. + if ((cap & cccr.card_cap_lsc) == 0 or (cap & cccr.card_cap_4bls) != 0) { + try cmd52Write(0, cccr.bus_width, cccr.bus_width_4); + setBusWidth(.four); + attachD3(); + } else { + state.width = .one; + } + } + + // ---- Block size 512 for function 0 and function 1, host side and card side. + try setCardBlockSize(0, io_block_size); + try setCardBlockSize(1, io_block_size); + setBlockSize(io_block_size); + + // ---- Finally the target frequency, now that the card is addressed and the bus is wide. + try setBusClock(state.target_khz); +} + +/// The 16-bit block size lives in two consecutive byte registers, low half first +/// (`port_esp_hosted_host_sdio.c:123-141`). Function n's copy is at `0x100 * n + 0x10`. +fn setCardBlockSize(func: u3, bytes: u16) Error!void { + const base: u17 = fbr_start * @as(u17, func); + try cmd52Write(0, base + cccr.blksize_l, @truncate(bytes)); + try cmd52Write(0, base + cccr.blksize_h, @truncate(bytes >> 8)); +} + +/// A bounded busy-wait, for the two places the SDIO specification asks for a delay between +/// retries. Same conservative frequency assumption as `Deadline`, in the same safe direction: on +/// this 90 MHz die a 10 ms request takes about 44 ms. +fn spinMicros(us: u32) void { + const d = Deadline.init(us); + while (!d.expired()) {} +} + +// ------------------------------------------------------------------------------------- CMD52 + +/// Read one byte from the card's register space. +/// +/// This is the whole minimal milestone: after `init` and `cardInit`, `cmd52Read(0, 0x00)` reads +/// CCCR offset 0 and the byte that comes back is the C6 answering. +pub fn cmd52Read(func: u3, addr: u17) Error!u8 { + const r = try sendCommand(.{ + .index = cmd_io_rw_direct, + .response = .short, + .check_crc = true, + }, cmd52Arg(false, func, addr, false, 0)); + return checkR5(r); +} + +/// Write one byte. The RAW flag is not set, matching `sdmmc_io_rw_direct` with SD_ARG_CMD52_WRITE +/// alone (`sdmmc_io.c:187`); `sdmmc_io_write_byte` adds SD_ARG_CMD52_EXCHANGE when it wants the +/// previous value back, which no caller here does. +pub fn cmd52Write(func: u3, addr: u17, value: u8) Error!void { + const r = try sendCommand(.{ + .index = cmd_io_rw_direct, + .response = .short, + .check_crc = true, + }, cmd52Arg(true, func, addr, false, value)); + _ = try checkR5(r); +} + +// ------------------------------------------------------------------------------------- CMD53 + +/// How one CMD53 is split. Two rules decide it, and both come from ESP-IDF rather than from the +/// SDIO specification, because both are properties of this controller: +/// +/// * **Block mode when the length is a whole number of 512-byte blocks**, byte mode otherwise. +/// In byte mode the count field is bytes and 0 encodes 512 ("See 5.3.1 SDIO simplified spec", +/// `sdmmc_io.c:351-355`), so one byte-mode command reaches 512 bytes and no further. +/// * **A byte-mode length of 4 or more must be a multiple of 4.** `sd_trans_sdmmc.c:526-532` +/// rejects anything else outright, and `sdmmc_io_read_bytes` works around it by splitting: +/// "host quirk: SDIO transfer with length not divisible by 4 bytes has to be split into two +/// transfers: one with aligned length, the other one for the remaining 1-3 bytes" +/// (`sdmmc_io.c:400-419`). So 6 bytes is two commands, 4 then 2, and 3 bytes is one. +/// +/// A caller that wants the split to be explicit - ESP-Hosted's block path does, because its +/// addresses increment across the split - can hand over one whole-block chunk at a time and get +/// exactly one block-mode command per call. A caller that does not can hand over any length. +const Chunk = struct { + block_mode: bool, + /// Bytes in this command. + len: u32, + /// The CMD53 count field: blocks in block mode, bytes in byte mode with 0 meaning 512. + count: u9, +}; + +fn nextChunk(remaining: u32) Chunk { + if (remaining >= io_block_size and remaining % io_block_size == 0) { + const max_blocks = bounce_len / io_block_size; + var blocks = remaining / io_block_size; + if (blocks > max_blocks) blocks = max_blocks; + return .{ + .block_mode = true, + .len = blocks * io_block_size, + .count = @intCast(blocks), + }; + } + var len = remaining; + if (len > io_block_size) len = io_block_size; + // The 4-byte rule. Below 4 bytes the whole request goes in one command; at or above it, the + // aligned part goes first and the 1-3 byte tail becomes the next chunk. + if (len >= 4 and len % 4 != 0) len &= ~@as(u32, 3); + return .{ + .block_mode = false, + .len = len, + .count = if (len == io_block_size) 0 else @intCast(len), + }; +} + +pub fn cmd53Read(func: u3, addr: u17, buf: []u8, incrementing: bool) Error!void { + var offset: u32 = 0; + var a: u32 = addr; + while (offset < buf.len) { + const c = nextChunk(@intCast(buf.len - offset)); + const arg = cmd53Arg(false, func, @truncate(a), c.block_mode, incrementing, c.count); + try dataTransfer(.read, arg, c.len, if (c.block_mode) io_block_size else c.len); + const dst = buf[offset..][0..c.len]; + const src = bufNc(); + for (dst, 0..) |*b, i| b.* = src[i]; + offset += c.len; + if (incrementing) a += c.len; + } +} + +pub fn cmd53Write(func: u3, addr: u17, data: []const u8, incrementing: bool) Error!void { + var offset: u32 = 0; + var a: u32 = addr; + while (offset < data.len) { + const c = nextChunk(@intCast(data.len - offset)); + const src = data[offset..][0..c.len]; + const dst = bufNc(); + for (src, 0..) |b, i| dst[i] = b; + // The IDMAC moves whole words, so a length that is not a multiple of 4 is rounded up + // (`sd_trans_sdmmc.c:127`). Zero the pad rather than send whatever the last transfer left. + var pad = c.len; + while (pad % 4 != 0) : (pad += 1) dst[pad] = 0; + const arg = cmd53Arg(true, func, @truncate(a), c.block_mode, incrementing, c.count); + try dataTransfer(.write, arg, c.len, if (c.block_mode) io_block_size else c.len); + offset += c.len; + if (incrementing) a += c.len; + } +} + +/// One CMD53 with its data phase, through the IDMAC and the bounce buffer. +/// +/// Order is ESP-IDF's (`sd_trans_sdmmc.c:524-568`): descriptor and transfer registers first, then +/// the command word, then wait for command-done and data-transfer-over in that order. Preparing +/// the DMA after starting the command would be a race against a card that answers immediately. +fn dataTransfer(dir: Direction, arg: u32, len: u32, blk: u32) Error!void { + std.debug.assert(len <= bounce_len); + + // The card must not still be holding DAT0 low from a previous write. + const busy = Deadline.init(busy_timeout_us); + while (status.get(data_busy) != 0) { + if (busy.expired()) return error.Busy; + } + + // As in `sendCommand`: the whole event set except this slot's card interrupt. `frun` is in + // `Event.data_errors` and nothing else ever clears it. + clearNonSlaveInterrupts(); + idsts.writeRaw(idsts_event_mask); + + const padded = (len + 3) & ~@as(u32, 3); + const d = descNc(); + d.buffer1 = busAddr(&dma.buf); + d.next = 0; + d.sizes = padded; // buffer1_size is [12:0]; buffer2 is unused + d.flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor | + Descriptor.last_descriptor | Descriptor.second_address_chained; + + setDataTransferLen(len); + setBlockSize(blk); + setDescriptorAddr(busAddr(&dma.desc)); + + // `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`, then the poll demand that tells the IDMAC to + // re-read a descriptor it may have parked on. + setDmaEnabled(true); + pldmnd.writeRaw(1); + + try startCommand(commandWord(.{ + .index = cmd_io_rw_extended, + .response = .short, + .check_crc = true, + .data = dir, + .slot = state.slot, + }), arg); + + _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us); + _ = try checkR5(resp0.raw()); + _ = try waitEvents(Event.dto, Event.data_errors, data_done_timeout_us); +} + +// -------------------------------------------------------------------- SDIO card interrupt (D1) + +/// Has the card asserted its interrupt line? +/// +/// Non-blocking, no side effect, straight out of RINTSTS bit 16+slot (`sdmmc_reg.h:621-631`). It +/// does **not** clear the bit; `clearSlaveInterrupt` does, deliberately, once a caller has decided +/// to act on it. +/// +/// Two different trigger behaviours meet at this bit and it is worth keeping them apart, because +/// conflating them sends you tuning the wrong knob: +/// +/// * **Card -> controller is an edge.** ESP-IDF: "SDIO interrupts are negedge sensitive ones: +/// the status bit is only set when first interrupt triggered" (`sd_host_sdmmc.c:396-402`). +/// That is why a waiter must check D1's level once before sleeping - an edge that arrived +/// while it was awake is not re-delivered. +/// * **Controller -> CLIC is a level.** RINTSTS is a sticky write-1-to-clear latch, so the +/// controller's output line stays asserted until software clears the bit that raised it. The +/// CLIC line therefore wants `.level`, and an edge trigger there would only hide a handler +/// that fails to deassert rather than fix it. +/// +/// This is the polling half. The interrupt half is a CLIC line and belongs to whoever owns the +/// scheduler: route `interrupt_source` with `hal.intr`, and in the handler mask the bit +/// (`setSlaveInterruptEnabled(false)`) before waking anybody. ESP-IDF does exactly that +/// (`sd_host_sdmmc.c:826-830`) and explains why at `:396-402`: "SDIO interrupts are negedge +/// sensitive ones: the status bit is only set when first interrupt triggered", so a handler that +/// leaves the bit unmasked and unhandled re-enters forever, and a waiter that sleeps without first +/// checking D1's level loses an edge that arrived while it was awake. +pub fn slaveInterruptPending() bool { + return rintsts.raw() & (Event.io_slot0 << state.slot) != 0; +} + +/// The raw masked-interrupt status word. Diagnostics only: a hang waiting on the card interrupt is +/// otherwise indistinguishable from a card that never asserted, and this is the register that tells +/// them apart. +pub fn interruptStatusRaw() u32 { + return rintsts.raw(); +} + +pub fn clearSlaveInterrupt() void { + rintsts.writeRaw(Event.io_slot0 << state.slot); +} + +/// INTMASK as written. The other half of "why is this line asserted": the controller's output is +/// RINTSTS AND INTMASK, and a diagnostic that prints only RINTSTS shows half the conjunction. +/// +/// **Not evidence about an arm.** INTMASK is an ordinary read/write register, so this returns +/// whatever the last store left - and on the interrupt path the last store is usually the *disarm*. +/// A caller that wants to know whether unmasking took effect must read the register back inside the +/// same masked region as the store; that is `armSlaveInterrupt`, and it exists because this +/// function was read as if it answered that question and it never could. +pub fn interruptMaskRaw() u32 { + return intmask.raw(); +} + +/// IDSTS - the IDMAC's own status word, which reaches the controller's interrupt output through +/// IDINTEN and *not* through INTMASK. +/// +/// The second independent reason the line can be asserted, and therefore the first thing to read +/// when a handler entry cannot be explained by RINTSTS. `muteInterrupts` leaves IDINTEN at zero so +/// this cannot raise the line in this driver; a foreign handler entry with bits set here means +/// something put IDINTEN back. +pub fn dmaStatusRaw() u32 { + return idsts.raw(); +} + +/// Clear every latched event *except* this slot's SDIO card interrupt. +/// +/// For an interrupt handler that has to lower the controller's output line without racing the +/// card: the card interrupt is the one event the handler is being woken for, and clearing it here +/// would drop the wakeup. Everything else - a stale command-done, a latched card-detect, an error +/// from a transfer that has already been reported - is safe to drop on the floor, and leaving any +/// of it latched while unmasked keeps the CLIC line high. +pub fn clearNonSlaveInterrupts() void { + rintsts.writeRaw(0x0003_ffff & ~(Event.io_slot0 << state.slot)); +} + +/// Unmask this slot's SDIO card interrupt, i.e. let it - and after `muteInterrupts`, *only* it - +/// reach the CLIC. RINTSTS records the event either way, so polling works without this. +/// +/// This is the whole of the masking half of arming a waiter: `muteInterrupts` has already left +/// every other bit of INTMASK and all of IDINTEN at zero, so `true` here makes this slot's card +/// interrupt the single reason the controller's output can assert - which is what a +/// level-triggered CLIC line with a one-cause handler requires. +/// +/// The *order* around it is the part that is easy to get wrong, and it belongs to whoever owns the +/// scheduler rather than here. `sd_host_slot_sdmmc_io_int_wait` (`sd_host_sdmmc.c:404-426`) is the +/// reference, and it is four steps: +/// +/// 1. `setSlaveInterruptEnabled(false)` - mask, so nothing arrives while the state is in flux. +/// 2. `clearSlaveInterrupt()` - drop the latched edge, so a stale one is not delivered as news. +/// 3. `slaveInterruptAsserted()` - **if true, act now and do not sleep.** The capture is a +/// negedge, so with D1 already low step 2 has just thrown away the only edge there will be. +/// 4. `setSlaveInterruptEnabled(true)` - unmask, and not before. Nothing can be lost between 2 +/// and 4: D1 is a level, and unmasking a bit RINTSTS has already latched asserts the line at +/// once. +/// +/// A handler on that line must mask again as its **unconditional first act**, on every path +/// including the one where the cause turns out not to be its own. The line is a level and it does +/// not lower itself. +pub fn setSlaveInterruptEnabled(on: bool) void { + // Masked, and that is not decoration. `sdioDispatch` performs *this same* read-modify-write on + // *this same* two-bit field, from an interrupt handler, as its unconditional first act. A task + // interrupted between the load and the store puts back the bit the handler had just cleared - + // re-arming a level-triggered line with nobody left waiting on it, which is precisely how the + // storm gets its second chance. Two CSR instructions, and `clkrst.Guard` composes: called from + // inside a handler, where MIE is already clear, it leaves MIE clear. + // + // The register writes are unchanged, so the `sdio_interrupt` differential case still compares + // the same resulting word against `sdmmc_ll_enable_sdio_interrupt`'s. + const guard = intr.mask(); + defer guard.release(); + const m = slotBit(); + const cur = intmask.get(sdio_int_mask); + intmask.modify(.{sdio_int_mask.is(if (on) cur | m else cur & ~m)}); +} + +/// What the controller reported the instant after this slot's card interrupt was unmasked. +pub const Armed = struct { + /// The bit the store was trying to set, i.e. `slaveInterruptMask()`. + want: u32, + /// INTMASK, read back inside the same masked region as the store. + intmask: u32, + /// MINTSTS - `RINTSTS & INTMASK`, and the only word the controller's output follows. Zero here + /// with `stuck()` true is the normal way to enter a sleep: the mask took and nothing is latched + /// yet. + mintsts: u32, + /// RINTSTS, for the case where the edge landed between the unmask and the read-back. + rintsts: u32, + + /// Did the unmask take effect? + pub inline fn stuck(self: Armed) bool { + return self.intmask & self.want != 0; + } +}; + +/// Unmask this slot's card interrupt and read the result back, both inside one masked region. +/// +/// This exists because the opposite conclusion was drawn from diagnostics that could not support +/// it. Every arming window on this board printed `intmask=0x00000000` and that was read as "the +/// unmask does not stick" - but `MARK PORT_SDIO_LAPSE` prints *after* the disarm, which had just +/// written that zero deliberately, and `interruptDiagnostics` runs on the application task, which +/// is never inside an arming window. Neither reading could ever have shown anything else, whatever +/// the hardware did. +/// +/// So the claim gets an instrument instead of an argument. Nothing runs between the store and the +/// three loads: no task, because the runtime is cooperative, and no handler, because MIE is clear. +/// A `stuck()` of false here is a fact about this register on this die; `stuck()` true retires the +/// hypothesis. +pub fn armSlaveInterrupt() Armed { + const guard = intr.mask(); + defer guard.release(); + const m = slotBit(); + intmask.modify(.{sdio_int_mask.is(intmask.get(sdio_int_mask) | m)}); + return .{ + .want = slaveInterruptMask(), + .intmask = intmask.raw(), + .mintsts = mintsts.raw(), + .rintsts = rintsts.raw(), + }; +} + +/// This slot's bit in RINTSTS/INTMASK/MINTSTS - the only interrupt cause a waiter here understands. +pub fn slaveInterruptMask() u32 { + return Event.io_slot0 << state.slot; +} + +/// Is the card asserting its interrupt *right now*? +/// +/// Read from D1's pad rather than from RINTSTS, because the two answer different questions: the +/// register says "a negedge was latched and not yet cleared", the pad says "the card is holding the +/// line low". Only the second is safe to test before sleeping, and it is what ESP-IDF tests - +/// `gpio_get_level(slot_ctx->io_config.d1_io) == 0` at `sd_host_sdmmc.c:413-415`. +/// +/// Requires D1's input buffer and matrix input to be configured, which `configurePins` does. +pub fn slaveInterruptAsserted() bool { + return gpio.getLevel(state.pins.d1) == 0; +} + +/// The masked status word - what the controller's interrupt output is actually looking at. +/// +/// Non-zero here and a silent CLIC means the delivery path above the controller is broken (source +/// routing, line enable, priority, threshold, mstatus.MIE). Zero here while `interruptStatusRaw` +/// is non-zero means the event is latched but masked, which is the normal resting state of this +/// driver. +pub fn interruptStatusMasked() u32 { + return mintsts.raw(); +} + +// ------------------------------------------------------------------------------- diagnostics + +/// `ets_printf` from the mask ROM, the same declaration `src/net/port.zig:75` makes and for the +/// same reason: this file's only module imports are `regs`, `mmio` and its sibling HALs, and the +/// symbol comes from the generated linker script rather than from any of them. +extern fn ets_printf(fmt: [*:0]const u8, ...) c_int; + +fn note(comptime fmt: [*:0]const u8, args: anytype) void { + _ = @call(.auto, ets_printf, .{fmt} ++ args); +} + +inline fn yesno(b: bool) u32 { + return @intFromBool(b); +} + +/// Print the whole card-interrupt delivery chain, in the order a signal traverses it, so that one +/// flash says which link is broken. Reads registers only: no loop, no wait, no side effect on any +/// of the state it reports. +/// +/// The chain has four links and each line below covers one: +/// +/// * `SDIO_DIAG_PAD` - the card's end. `asserted=1` means D1 is low, i.e. the C6 is requesting +/// service at this instant. `ie=0` or `in_src` not equal to D1's pad number means the +/// controller cannot see D1 at all, and no amount of unmasking will help. +/// * `SDIO_DIAG_TIEOFF` - the two matrix inputs with no pin on this board. `card_int_n` must read +/// 63 (constant one, inactive) and `card_detect_n` 62 (constant zero, card present), both with +/// `from_matrix=1`. A `card_int_n` stuck at a constant *zero* is an interrupt that is asserted +/// before software ever runs, so the first negedge happens before anyone is watching and no +/// second one ever comes. +/// * `SDIO_DIAG_CTLR` - the controller's end. `latched` is RINTSTS's bit for this slot, +/// `unmasked` is INTMASK's, and `mintsts` is the conjunction the interrupt output follows. +/// `idsts`/`idinten` are the other, independent reason this output can be asserted. +/// +/// **`unmasked=0` here is the resting state and is not a finding.** This function is called +/// from an application task; the card interrupt is unmasked only inside an arming window, on +/// the transport's own task, and is masked again by the handler or the disarm before that task +/// yields. So an application can never observe the mask up, whatever the hardware does, and +/// reading a zero here as "the unmask does not stick" is what cost this path a week. The +/// register read that can answer that question is `armSlaveInterrupt`. +/// * `SDIO_DIAG_CLIC` - delivery. An unrouted source, a clear `enabled`, a priority at or below +/// `thresh`, or `mie=0` each mean the line exists and cannot arrive. +pub fn interruptDiagnostics() void { + const m = slaveInterruptMask(); + const rsts = rintsts.raw(); + const imask = intmask.raw(); + const d1 = state.pins.d1; + const d1_in = gpio.matrixInSource(sig.cdata1); + const ci = gpio.matrixInSource(sig.card_int); + const cdet = gpio.matrixInSource(sig.card_detect); + + note("MARK SDIO_DIAG_PAD d1=gpio%u level=%u asserted=%u ie=%u in_src=%u from_matrix=%u inv=%u expect in_src=%u\r\n", .{ + @as(u32, d1), + @as(u32, gpio.getLevel(d1)), + yesno(slaveInterruptAsserted()), + yesno(gpio.isInputEnabled(d1)), + @as(u32, d1_in.pin), + yesno(d1_in.from_matrix), + yesno(d1_in.inverted), + @as(u32, d1), + }); + note("MARK SDIO_DIAG_TIEOFF card_int_n=%u/%u card_detect_n=%u/%u expect 63/1 and 62/1\r\n", .{ + @as(u32, ci.pin), yesno(ci.from_matrix), + @as(u32, cdet.pin), yesno(cdet.from_matrix), + }); + note("MARK SDIO_DIAG_CTLR slot=%u bit=0x%05x latched=%u unmasked=%u rintsts=0x%08x intmask=0x%08x mintsts=0x%08x\r\n", .{ + @as(u32, state.slot), + m, + yesno(rsts & m != 0), + yesno(imask & m != 0), + rsts, + imask, + interruptStatusMasked(), + }); + note("MARK SDIO_DIAG_CTRL ctrl=0x%08x int_enable=%u idsts=0x%08x idinten=0x%08x status=0x%08x clkena=0x%08x\r\n", .{ + ctrl.raw(), + ctrl.get(int_enable), + idsts.raw(), + idinten.raw(), + status.raw(), + clkena.raw(), + }); + + const src: u32 = @intFromEnum(interrupt_source); + if (intr.routedLine(interrupt_source)) |line| { + note("MARK SDIO_DIAG_CLIC source=%u line=%u enabled=%u pending=%u trigger=%u prio=%u thresh=%u mie=%u\r\n", .{ + src, + @as(u32, line), + yesno(intr.isEnabled(line)), + yesno(intr.isPending(line)), + @as(u32, @intFromEnum(intr.getTrigger(line))), + @as(u32, intr.getPriority(line)), + @as(u32, intr.getThreshold()), + yesno(intr.globalEnabled()), + }); + } else { + note("MARK SDIO_DIAG_CLIC source=%u UNROUTED - no CLIC line can deliver this interrupt\r\n", .{src}); + } +} + +// --------------------------------------------------------------------------------- host tests +// +// Everything below runs on the host under `zig build test`. It covers the two things in this file +// that are pure functions of their arguments - the command word and the CMD52/CMD53 argument +// layouts - plus the divider table and the chunking rule. The register sequences are not testable +// here; that is what src/oracle/sdmmc_cases.zig is for. + +const testing = std.testing; + +test "CMD52 read is the word ESP-IDF builds" { + // make_hw_cmd for {opcode 52, SCF_CMD_AC | SCF_RSP_R5}: response_expect (R5 is PRESENT), + // check_response_crc (R5 has CRC), wait_complete (not CMD0/12/11), no data. Then + // sd_host_slot_start_command adds use_hold_reg, card_num and start_command. + const w = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_2174), w); +} + +test "CMD52 on slot 0 differs from slot 1 only in card_num" { + const s0 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 0 }); + const s1 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA000_2174), s0); + try testing.expectEqual(@as(u32, 1 << 16), s0 ^ s1); +} + +test "CMD53 sets data_expected, and rw only when writing" { + const rd = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .read, .slot = 1 }); + const wr = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .write, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_2375), rd); + try testing.expectEqual(@as(u32, 0xA001_2775), wr); + try testing.expectEqual(@as(u32, 1 << 10), rd ^ wr); +} + +test "CMD0 sends the init sequence and expects nothing back" { + // The only command make_hw_cmd gives send_init and denies wait_complete. + const w = commandWord(.{ .index = 0, .send_init = true, .wait_prvdata = false, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_8000), w); + try testing.expectEqual(@as(u32, 0), w & (1 << 6)); // no response expected +} + +test "CMD5's response CRC is not checked" { + // R4 is SCF_RSP_PRESENT alone (sd_protocol_types.h:141) - the OCR response carries no valid + // CRC7, and checking it would fail every card. + const w = commandWord(.{ .index = 5, .response = .short, .check_crc = false, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_2045), w); + try testing.expectEqual(@as(u32, 0), w & (1 << 8)); +} + +test "CMD3 and CMD7 do check it" { + try testing.expectEqual( + @as(u32, 0xA001_2143), + commandWord(.{ .index = 3, .response = .short, .check_crc = true, .slot = 1 }), + ); + try testing.expectEqual( + @as(u32, 0xA001_2147), + commandWord(.{ .index = 7, .response = .short, .check_crc = true, .slot = 1 }), + ); +} + +test "the clock update command sends nothing to the card" { + const w = commandWord(.{ .index = 0, .update_clock = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA021_2000), w); + try testing.expectEqual(@as(u32, 1 << 21), w & (1 << 21)); + try testing.expectEqual(@as(u32, 0), w & (1 << 6)); +} + +test "a long response sets response_length as well as response_expect" { + const w = commandWord(.{ .index = 2, .response = .long, .check_crc = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 1 << 7), w & (1 << 7)); + try testing.expectEqual(@as(u32, 1 << 6), w & (1 << 6)); +} + +test "every command word starts the command and uses the hold register" { + for ([_]Command{ + .{ .index = 52, .response = .short, .check_crc = true }, + .{ .index = 53, .response = .short, .check_crc = true, .data = .read }, + .{ .index = 0, .send_init = true, .wait_prvdata = false }, + }) |c| { + const w = commandWord(c); + try testing.expect(w & (1 << 31) != 0); + try testing.expect(w & (1 << 29) != 0); + } +} + +test "CMD52 argument layout" { + // Read CCCR 0x00 on function 0: everything zero. + try testing.expectEqual(@as(u32, 0), cmd52Arg(false, 0, 0x00, false, 0)); + // Write 0x02 to CCCR 0x02 (I/O enable) on function 0. + try testing.expectEqual(@as(u32, 0x8000_0402), cmd52Arg(true, 0, 0x02, false, 0x02)); + // Function 1, address 0x1F800, data 0xAB, with the read-after-write flag. + const a = cmd52Arg(true, 1, 0x1F800, true, 0xAB); + try testing.expectEqual(@as(u32, 1), a >> 31); + try testing.expectEqual(@as(u32, 1), (a >> 28) & 0x7); + try testing.expectEqual(@as(u32, 1), (a >> 27) & 1); + try testing.expectEqual(@as(u32, 0x1F800), (a >> 9) & 0x1FFFF); + try testing.expectEqual(@as(u32, 0xAB), a & 0xFF); +} + +test "CMD53 argument layout, both modes" { + // Block mode, function 1, address 0, incrementing, one block. + const blk = cmd53Arg(false, 1, 0, true, true, 1); + try testing.expectEqual(@as(u32, 0x1C00_0001), blk); + // Byte mode, function 1, fixed address, 12 bytes - the ESP-Hosted length read. + const byt = cmd53Arg(false, 1, 0x058, false, false, 12); + try testing.expectEqual(@as(u32, 0x1000_B00C), byt); + // Writing sets bit 31 and nothing else. + try testing.expectEqual( + @as(u32, 1) << 31, + cmd53Arg(true, 1, 0x058, false, false, 12) ^ byt, + ); +} + +test "byte mode encodes 512 as a count of zero" { + // SDIO simplified spec 5.3.1, as applied at sdmmc_io.c:351-355. The chunker never produces + // this case - a 512-byte request is a whole block and goes block mode - so the encoder is + // checked directly. ESP-Hosted can still reach it: a 512-byte read at a fixed address. + try testing.expectEqual(@as(u32, 0), cmd53Arg(false, 1, 0, false, true, 0) & 0x1ff); + const c = nextChunk(512); + try testing.expect(c.block_mode); + try testing.expectEqual(@as(u32, 512), c.len); + try testing.expectEqual(@as(u9, 1), c.count); +} + +test "chunking splits on block boundaries and clamps to the bounce buffer" { + // Whole blocks, within the buffer: one block-mode command. + try testing.expectEqual(@as(u32, 1024), nextChunk(1024).len); + try testing.expect(nextChunk(1024).block_mode); + // More blocks than fit: clamped to bounce_len, still block mode, still whole blocks. + const big = nextChunk(8192); + try testing.expect(big.block_mode); + try testing.expectEqual(bounce_len, big.len); + try testing.expectEqual(@as(u9, bounce_len / 512), big.count); + // Not a block multiple: byte mode, count in bytes. + const odd = nextChunk(12); + try testing.expect(!odd.block_mode); + try testing.expectEqual(@as(u32, 12), odd.len); + try testing.expectEqual(@as(u9, 12), odd.count); + // Longer than one byte-mode command can carry: clamped to 512. + const long = nextChunk(1000); + try testing.expect(!long.block_mode); + try testing.expectEqual(@as(u32, 512), long.len); +} + +test "the controller's 4-byte rule turns a 6-byte transfer into 4 then 2" { + // sd_trans_sdmmc.c:526-532 rejects a length that is >= 4 and not a multiple of 4 outright. + const first = nextChunk(6); + try testing.expect(!first.block_mode); + try testing.expectEqual(@as(u32, 4), first.len); + const second = nextChunk(6 - first.len); + try testing.expectEqual(@as(u32, 2), second.len); + // Under four bytes the whole thing goes in one command; that is the case the rule exempts. + try testing.expectEqual(@as(u32, 3), nextChunk(3).len); + try testing.expectEqual(@as(u32, 1), nextChunk(1).len); + // And every chunk a loop produces is either aligned or a final short tail. + var remaining: u32 = 1023; + var commands: u32 = 0; + while (remaining > 0) { + const c = nextChunk(remaining); + try testing.expect(c.len > 0); + try testing.expect(c.len < 4 or c.len % 4 == 0); + remaining -= c.len; + commands += 1; + try testing.expect(commands < 8); // 512 + 508 + 3, not an unbounded walk + } +} + +test "divider table reproduces ESP-IDF's three named frequencies" { + try testing.expectEqual(Dividers{ .host = 10, .card = 20 }, dividersFor(400)); + try testing.expectEqual(Dividers{ .host = 8, .card = 0 }, dividersFor(20_000)); + try testing.expectEqual(Dividers{ .host = 4, .card = 0 }, dividersFor(40_000)); +} + +test "divider table lands on or below the requested frequency" { + for ([_]u32{ 400, 1_000, 5_000, 10_000, 20_000, 25_000, 40_000 }) |khz| { + const d = dividersFor(khz); + const div: u64 = @as(u64, d.host) * (if (d.card == 0) @as(u64, 1) else @as(u64, d.card) * 2); + const actual_khz = 160_000 / div; + try testing.expect(actual_khz <= khz); + } +} + +test "the DMA region is one aligned block of exactly the documented size" { + try testing.expectEqual(@as(usize, 16), @sizeOf(Descriptor)); + try testing.expectEqual(@as(usize, 64 + bounce_len), @sizeOf(DmaRegion)); + try testing.expectEqual(@as(usize, 0), @offsetOf(DmaRegion, "desc")); + try testing.expectEqual(@as(usize, 64), @offsetOf(DmaRegion, "buf")); + // Both halves of what the one-shot cache maintenance call needs: a base on a cache line and a + // length that is a whole number of them. The alignment is on the variable, not on the type - + // `@alignOf(DmaRegion)` is 4 - so it has to be checked on the object. + try testing.expectEqual(@as(usize, 0), @intFromPtr(&dma) % cache_line); + try testing.expectEqual(@as(usize, 0), @sizeOf(DmaRegion) % cache_line); +} + +test "the non-cacheable alias is a fixed offset and nothing more" { + var cell: u32 = 0; + const a = cachedAddr(&cell); + try testing.expectEqual(a +% @as(usize, 0x4000_0000), uncachedAddr(&cell)); + // And it is the offset ESP-IDF uses, not one this file invented. + try testing.expectEqual(@as(u32, 0x4000_0000), non_cacheable_offset); +} + +test "descriptor flags are the bits sdmmc_struct.h names" { + try testing.expectEqual(@as(u32, 1 << 2), Descriptor.last_descriptor); + try testing.expectEqual(@as(u32, 1 << 3), Descriptor.first_descriptor); + try testing.expectEqual(@as(u32, 1 << 4), Descriptor.second_address_chained); + try testing.expectEqual(@as(u32, 1 << 31), Descriptor.owned_by_idmac); + // The word a single-descriptor transfer writes. + const flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor | + Descriptor.last_descriptor | Descriptor.second_address_chained; + try testing.expectEqual(@as(u32, 0x8000_001C), flags); +} + +test "the default interrupt mask is ESP-IDF's SDMMC_LL_EVENT_DEFAULT" { + // sdmmc_ll.h:64-69, expanded: CD|RESP_ERR|CMD_DONE|DATA_OVER|RCRC|DCRC|RTO|DTO|HTO|HLE|SBE|EBE + try testing.expectEqual(@as(u32, 0xB7CF), Event.default); + // and it deliberately excludes the two per-FIFO-word requests and both SDIO card interrupts. + try testing.expectEqual(@as(u32, 0), Event.default & (Event.txdr | Event.rxdr)); + try testing.expectEqual(@as(u32, 0), Event.default & (Event.io_slot0 | Event.io_slot1)); +} + +test "the armed mask drops card detect, and nothing else" { + // The bit that produced an unstoppable CLIC line 21: cd latches during pin setup, nothing in + // the command path clears it, and while it is unmasked the controller's output never + // deasserts. `configureInterrupts` writes `armed`, not `default`. + try testing.expectEqual(@as(u32, 0xB7CE), Event.armed); + try testing.expectEqual(@as(u32, 0), Event.armed & Event.cd); + try testing.expectEqual(Event.cd, Event.default ^ Event.armed); + // Every event a transfer actually waits on survives the change. + for ([_]u32{ Event.cmd_done, Event.dto, Event.re, Event.rcrc, Event.dcrc, Event.rto, Event.drto, Event.hto, Event.hle, Event.sbe, Event.ebe }) |e| { + try testing.expect(Event.armed & e != 0); + } +} diff --git a/src/hal/systimer.zig b/src/hal/systimer.zig new file mode 100644 index 0000000..dc316bf --- /dev/null +++ b/src/hal/systimer.zig @@ -0,0 +1,151 @@ +//! SYSTIMER: two 52-bit counters on a fixed clock, plus three comparators each. +//! +//! This is the most useful peripheral on the chip for bring-up work and the cheapest to trust. Its +//! source is fixed - XTAL at 40 MHz, divided to 16 MHz (`clk_tree_defs.h:196-198`) - so unlike the +//! CPU cycle counter its rate does not move when the clock tree is reconfigured, and unlike the +//! timer groups it needs no divider arithmetic and no pads. +//! +//! Reading it is a **sequence**, not a load, and that is the interesting part: +//! +//! write UNIT0_UPDATE = 1 -> ask the peripheral to latch its counter +//! poll UNIT0_VALUE_VALID -> wait for the latch +//! read VALUE_HI, then VALUE_LO -> read the latched pair +//! +//! Skip the handshake and you read a value that is being incremented underneath you: the low word +//! can wrap between the two loads, so `hi` belongs to one instant and `lo` to the next, and the +//! result jumps backwards by 2^32 ticks about once every 268 seconds at 16 MHz. A register +//! snapshot taken after either version looks identical - which is exactly why the differential +//! harness records the *write trace* as well as the final state. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// Ticks per second. XTAL/2.5 = 16 MHz, fixed: `SYSTIMER_CLK_SRC_XTAL` with the divider ESP-IDF +/// programs in `systimer_hal_init`. Not derived from the CPU clock, which on this board is whatever +/// the bootloader left (measured ~90 MHz, not the 360 the part is rated for). +pub const hz: u32 = 16_000_000; + +const conf = Reg.at(regs.SYSTIMER_CONF_REG); +const clk_en = Field.of(regs.SYSTIMER_CLK_EN_S, regs.SYSTIMER_CLK_EN_V); + +/// The two counter units. `unit_op` holds the update/valid handshake bits, `value_hi`/`value_lo` the +/// latched result. Strides are derived from consecutive macros, not assumed. +const unit_op = mmio.RegArray(regs.SYSTIMER_UNIT0_OP_REG, regs.SYSTIMER_UNIT1_OP_REG, 2); +const unit_value_hi = mmio.RegArray(regs.SYSTIMER_UNIT0_VALUE_HI_REG, regs.SYSTIMER_UNIT1_VALUE_HI_REG, 2); +const unit_value_lo = mmio.RegArray(regs.SYSTIMER_UNIT0_VALUE_LO_REG, regs.SYSTIMER_UNIT1_VALUE_LO_REG, 2); + +// The per-unit fields split into two groups, and the split is not obvious from the names. +// +// `update` and `valid` live in a *per-unit* register (UNIT0_OP_REG, UNIT1_OP_REG) and therefore sit +// at the same bit in each - asserted below, so indexing the register is enough. +// +// `work_en` is different: both units' enables live in the *shared* SYSTIMER_CONF_REG, at bits 30 and +// 29 respectively. A first draft of this file used unit 0's field for both, which would have enabled +// the wrong counter and left the requested one dead; the comptime assert caught it before it ever +// reached the chip. Hence a per-unit lookup rather than one constant. +const update = Field.of(regs.SYSTIMER_TIMER_UNIT0_UPDATE_S, regs.SYSTIMER_TIMER_UNIT0_UPDATE_V); +const valid = Field.of(regs.SYSTIMER_TIMER_UNIT0_VALUE_VALID_S, regs.SYSTIMER_TIMER_UNIT0_VALUE_VALID_V); +const value_hi = Field.of(regs.SYSTIMER_TIMER_UNIT0_VALUE_HI_S, regs.SYSTIMER_TIMER_UNIT0_VALUE_HI_V); + +comptime { + const update1 = Field.of(regs.SYSTIMER_TIMER_UNIT1_UPDATE_S, regs.SYSTIMER_TIMER_UNIT1_UPDATE_V); + const valid1 = Field.of(regs.SYSTIMER_TIMER_UNIT1_VALUE_VALID_S, regs.SYSTIMER_TIMER_UNIT1_VALUE_VALID_V); + if (update1.shift != update.shift or valid1.shift != valid.shift) + @compileError("the systimer units' OP registers disagree on bit positions; index per unit"); + // The other half of the same story: these two MUST differ, because they share a register. + if (workEn(.unit0).shift == workEn(.unit1).shift) + @compileError("both work_en fields claim the same bit of SYSTIMER_CONF; one macro is wrong"); +} + +inline fn workEn(comptime unit: Unit) Field { + return switch (unit) { + .unit0 => Field.of(regs.SYSTIMER_TIMER_UNIT0_WORK_EN_S, regs.SYSTIMER_TIMER_UNIT0_WORK_EN_V), + .unit1 => Field.of(regs.SYSTIMER_TIMER_UNIT1_WORK_EN_S, regs.SYSTIMER_TIMER_UNIT1_WORK_EN_V), + }; +} + +pub const Unit = enum(u1) { unit0 = 0, unit1 = 1 }; + +/// The counter's own clock gate, inside the peripheral and separate from the bus clock gate in +/// HP_SYS_CLKRST. +pub fn setEnabled(on: bool) void { + conf.modify(.{clk_en.is(@intFromBool(on))}); +} + +pub fn setUnitEnabled(comptime unit: Unit, on: bool) void { + conf.modify(.{workEn(unit).is(@intFromBool(on))}); +} + +/// Bring the peripheral up: bus clock and reset through CLKRST, then its internal gate and unit. +/// +/// Deliberately does *not* reprogram the clock source or divider. The bootloader has already set +/// those, ESP-IDF's own `systimer_hal_init` would set them the same way, and re-running that on a +/// live counter makes the timebase jump - which would corrupt any measurement taken across the call. +pub fn init() void { + clkrst.setClockEnabled(.systimer, true); + setEnabled(true); + setUnitEnabled(.unit0, true); +} + +/// The 52-bit counter, latched through the update/valid handshake. +/// +/// Returns null if the peripheral does not acknowledge within `spins` reads, rather than spinning +/// forever: a systimer whose clock is gated off never sets `valid`, and hanging in a HAL call with +/// no output is the worst possible way to report that. +pub fn read(unit: Unit) ?u64 { + const i: u32 = @intFromEnum(unit); + const op = unit_op.at(i); + + // Ask for a snapshot, and clear the previous handshake in the same store. + // + // This has to be a read-modify-write, and a whole-word `write` is a bug. UPDATE is bit 30 and + // `WT`, so writing it as a single store looks right - but VALUE_VALID is bit 29 of the same word + // and is `R/SS/WTC`, write-1-to-clear (systimer_reg.h). A whole-word store writes 0 there, which + // is the no-op for a W1C bit, so the valid flag from the *previous* snapshot is never cleared: + // after one successful read it stays set forever, the poll below exits immediately on a stale + // flag, and the HI/LO pair that follows can straddle two different snapshots - precisely the + // tearing this handshake exists to prevent. + // + // ESP-IDF gets this right by accident of its idiom: `systimer_ll_counter_snapshot` assigns a + // bitfield of a `volatile` union, which compiles to a 32-bit read-modify-write that writes bit + // 29 back as 1 whenever it read 1, clearing it and re-arming in one store. This does the same + // thing deliberately. + // + // The register differential cannot see this: once any snapshot has completed, UNIT0_OP reads + // 0x2000_0000 under either version. + op.writeRaw(op.raw() | update.mask()); + + var spins: u32 = 0; + while (op.get(valid) == 0) { + spins += 1; + if (spins > 10_000) return null; + } + + // Order matters less than the latch does - both words are frozen now - but read high first to + // match ESP-IDF's LL, so the write/read trace lines up under differential test. + const hi: u64 = unit_value_hi.at(i).get(value_hi); + const lo: u64 = unit_value_lo.at(i).raw(); + return (hi << 32) | lo; +} + +/// Microseconds since the counter started, from the 16 MHz tick. +pub fn micros(unit: Unit) ?u64 { + const ticks = read(unit) orelse return null; + return ticks / (hz / 1_000_000); +} + +/// Busy-wait. Uses the counter rather than the CPU cycle count, so the delay is right regardless of +/// what the CPU clock happens to be. +pub fn delayMicros(us: u32) void { + const start = read(.unit0) orelse return; + const target = start + @as(u64, us) * (hz / 1_000_000); + while (true) { + const now = read(.unit0) orelse return; + if (now >= target) return; + } +} diff --git a/src/hal/timg.zig b/src/hal/timg.zig new file mode 100644 index 0000000..f669387 --- /dev/null +++ b/src/hal/timg.zig @@ -0,0 +1,513 @@ +//! The timer groups: TIMG0 and TIMG1, each two general-purpose 54-bit timers plus one MWDT. +//! +//! Three unrelated functions share one register block (timg_ll.h:7 says so in as many words): +//! the general-purpose timers, the main watchdog, and RTC clock calibration. Only the first two are +//! here; calibration belongs to the clock tree, and ETM and interrupts are deliberately absent. +//! +//! Four things about this block cost real care, all of them taken from ESP-IDF's LL rather than +//! guessed at: +//! +//! **Reading the counter is a sequence, not a load** (timer_ll.h:248-269). The counter lives in a +//! different clock domain from the register file, so its value only appears in `TxLO`/`TxHI` after a +//! software capture: +//! +//! write TIMG_TxUPDATE = 1 -> ask for a capture +//! poll until TIMG_Tx_UPDATE == 0 -> the hardware clears it when the pair is latched +//! read TxHI, then TxLO -> 22 bits + 32 bits = the 54-bit count +//! +//! Note the polarity: unlike SYSTIMER, which sets a separate `VALUE_VALID` bit, this peripheral +//! *clears the request bit* to acknowledge. Waiting for it to become 1 hangs forever; not waiting at +//! all returns whatever the last capture left, which for a never-captured timer is 0 and therefore +//! looks like a stopped timer rather than like a bug. +//! +//! **The watchdog registers are write-protected, and the key is the reset value** (mwdt_ll.h:231-244 +//! and timer_group_reg.h, TIMG_WDT_WKEY: "If the register contains a different value than its reset +//! value, write protection is enabled", default 1356348065 = 0x50D83AA1). So "unlock" means writing +//! the key back, and "lock" means writing anything else - IDF writes 0. A watchdog register write +//! made while locked is silently dropped, which is the failure mode this file's API shape exists to +//! prevent: every MWDT operation is a method on the `Watchdog` handle returned by `unlock`, and +//! there is no way to reach one without holding it: +//! +//! const wdt = timg.unlock(.timg1); +//! defer wdt.release(); +//! wdt.setStage(.stage0, 2_000_000, .reset_system); +//! +//! **Watchdog configuration is committed asynchronously.** Every write to WDTCONFIG0-5 has to be +//! followed by `WDT_CONF_UPDATE_EN` (mwdt_ll.h:122, and again after every other config write), which +//! is a write-to-trigger bit. The exception is `WDT_EN` itself: `mwdt_ll_enable`/`_disable` +//! (mwdt_ll.h:61-77) do *not* pulse it, so neither does `setEnabled` - matching IDF exactly matters +//! more here than consistency, because the differential harness compares the resulting word. +//! +//! **Do not resurrect a watchdog you are not feeding.** TIMG0 hosts MWDT0, which this image's +//! bootloader has already disabled, and `TIMG_WDT_FLASHBOOT_MOD_EN` defaults to 1 and runs the +//! watchdog *independently of* `WDT_EN` (mwdt_ll.h:186-188). Resetting a timer group therefore +//! re-arms flash-boot protection and reboots the board a moment later with nothing on the console to +//! explain it; `clkrst.resetPeripheral` clears the bit as part of the reset for exactly this reason +//! (its `clears_flashboot` flag), which is why nothing in this file pulses a reset bit itself. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// TIMG_LL_INST_NUM (timg_ll.h:20). +pub const group_count = 2; +/// TIMG_LL_GPTIMERS_PER_INST (timg_ll.h:23). Two per group on the P4, unlike the C-series parts. +pub const timers_per_group = 2; +/// TIMER_LL_COUNTER_BIT_WIDTH (timer_ll.h:25). 32 bits in `TxLO` plus 22 in `TxHI`. +pub const counter_bits = 54; + +pub const Group = enum(u1) { timg0 = 0, timg1 = 1 }; +pub const Timer = enum(u1) { t0 = 0, t1 = 1 }; + +// -------------------------------------------------------------------------------- addressing +// +// The macros are indexed two different ways at once and neither is derivable from the other: +// `TIMG_T0CONFIG_REG(i)` takes the *group*, while the *timer* is baked into the macro name +// (`T0CONFIG` vs `T1CONFIG`). Rather than duplicate every accessor per timer, the timer index is +// turned into a stride - but a stride assumed is a stride that eventually writes into the next +// register, so both strides are checked at comptime against the macros for the other instance. + +const group_stride = mmio.addr(regs.TIMG_T0CONFIG_REG(1)) - mmio.addr(regs.TIMG_T0CONFIG_REG(0)); +const timer_stride = mmio.addr(regs.TIMG_T1CONFIG_REG(0)) - mmio.addr(regs.TIMG_T0CONFIG_REG(0)); + +// Absolute addresses of group 0 / timer 0's registers. Every other (group, timer) is these plus a +// multiple of the two strides. +const a_config = mmio.addr(regs.TIMG_T0CONFIG_REG(0)); +const a_lo = mmio.addr(regs.TIMG_T0LO_REG(0)); +const a_hi = mmio.addr(regs.TIMG_T0HI_REG(0)); +const a_update = mmio.addr(regs.TIMG_T0UPDATE_REG(0)); +const a_alarm_lo = mmio.addr(regs.TIMG_T0ALARMLO_REG(0)); +const a_alarm_hi = mmio.addr(regs.TIMG_T0ALARMHI_REG(0)); +const a_load_lo = mmio.addr(regs.TIMG_T0LOADLO_REG(0)); +const a_load_hi = mmio.addr(regs.TIMG_T0LOADHI_REG(0)); +const a_load = mmio.addr(regs.TIMG_T0LOAD_REG(0)); + +comptime { + // The timer sub-block is contiguous and uniform - assert it, per register, rather than trust + // that 0x24 happens to be right for all nine. + const pairs = .{ + .{ a_config, mmio.addr(regs.TIMG_T1CONFIG_REG(0)) }, + .{ a_lo, mmio.addr(regs.TIMG_T1LO_REG(0)) }, + .{ a_hi, mmio.addr(regs.TIMG_T1HI_REG(0)) }, + .{ a_update, mmio.addr(regs.TIMG_T1UPDATE_REG(0)) }, + .{ a_alarm_lo, mmio.addr(regs.TIMG_T1ALARMLO_REG(0)) }, + .{ a_alarm_hi, mmio.addr(regs.TIMG_T1ALARMHI_REG(0)) }, + .{ a_load_lo, mmio.addr(regs.TIMG_T1LOADLO_REG(0)) }, + .{ a_load_hi, mmio.addr(regs.TIMG_T1LOADHI_REG(0)) }, + .{ a_load, mmio.addr(regs.TIMG_T1LOAD_REG(0)) }, + }; + for (pairs) |p| { + if (p[1] - p[0] != timer_stride) @compileError( + "the two timers' registers are not a uniform stride apart; index them per timer", + ); + } + // And the group stride is the same for a register other than CONFIG. + if (mmio.addr(regs.TIMG_T0LO_REG(1)) - a_lo != group_stride) + @compileError("the two timer groups are not a uniform stride apart"); + + // T0's and T1's *fields* sit at the same bit positions in their respective registers, which is + // what makes one set of Field constants enough. If a future register set moves one of them, + // this stops the build instead of writing the divider into the alarm enable. + const t1_divider = Field.of(regs.TIMG_T1_DIVIDER_S, regs.TIMG_T1_DIVIDER_V); + const t1_en = Field.of(regs.TIMG_T1_EN_S, regs.TIMG_T1_EN_V); + const t1_update = Field.of(regs.TIMG_T1_UPDATE_S, regs.TIMG_T1_UPDATE_V); + const t1_hi = Field.of(regs.TIMG_T1_HI_S, regs.TIMG_T1_HI_V); + if (t1_divider.shift != divider.shift or t1_divider.width != divider.width or + t1_en.shift != counter_en.shift or t1_update.shift != update.shift or + t1_hi.width != count_hi.width) + @compileError("timer 0 and timer 1 disagree on field positions; look up fields per timer"); +} + +inline fn tReg(comptime a0: u32, g: Group, t: Timer) Reg { + return Reg.atAddress(a0 + + group_stride * @as(u32, @intFromEnum(g)) + + timer_stride * @as(u32, @intFromEnum(t))); +} + +/// `a0` is not comptime: the stage-timeout registers are picked by a runtime `Stage` +/// (`stageHoldAddr`), and every other caller passes a constant that folds anyway. +inline fn gReg(a0: u32, g: Group) Reg { + return Reg.atAddress(a0 + group_stride * @as(u32, @intFromEnum(g))); +} + +// TxCONFIG fields. `divcnt_rst` is write-to-trigger; the rest are plain R/W. +const alarm_en = Field.of(regs.TIMG_T0_ALARM_EN_S, regs.TIMG_T0_ALARM_EN_V); +const divcnt_rst = Field.of(regs.TIMG_T0_DIVCNT_RST_S, regs.TIMG_T0_DIVCNT_RST_V); +const divider = Field.of(regs.TIMG_T0_DIVIDER_S, regs.TIMG_T0_DIVIDER_V); +const autoreload = Field.of(regs.TIMG_T0_AUTORELOAD_S, regs.TIMG_T0_AUTORELOAD_V); +const increase = Field.of(regs.TIMG_T0_INCREASE_S, regs.TIMG_T0_INCREASE_V); +const counter_en = Field.of(regs.TIMG_T0_EN_S, regs.TIMG_T0_EN_V); +const update = Field.of(regs.TIMG_T0_UPDATE_S, regs.TIMG_T0_UPDATE_V); +const count_hi = Field.of(regs.TIMG_T0_HI_S, regs.TIMG_T0_HI_V); +const alarm_value_hi = Field.of(regs.TIMG_T0_ALARM_HI_S, regs.TIMG_T0_ALARM_HI_V); +const load_value_hi = Field.of(regs.TIMG_T0_LOAD_HI_S, regs.TIMG_T0_LOAD_HI_V); + +// ------------------------------------------------------------------------------ timer clocks +// +// The timers' function clock is selected and gated in HP_SYS_CLKRST, not in the timer group: group 0 +// in PERI_CLK_CTRL20 and group 1 in PERI_CLK_CTRL21 (timer_ll.h:117-129, :146-160). Two shared +// registers, so both operations take the interrupt guard - the same read-modify-write hazard +// `clkrst` exists for. + +const peri_clk_ctrl20 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL20_REG); +const peri_clk_ctrl21 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL21_REG); + +/// The three function clocks a GP timer can run from, with the encodings from +/// `timer_ll_set_clock_source` (timer_ll.h:100-116). The numbering is not the enum order anyone +/// would pick: XTAL is 0, RC_FAST is 1, PLL_F80M is 2. +pub const ClockSource = enum(u2) { + xtal = 0, + rc_fast = 1, + pll_f80m = 2, +}; + +/// Where the group/timer's source-select and gate fields live. Both are in one word per group, and +/// the bit positions differ per timer, so this is a genuine per-instance lookup rather than a stride. +const TimerClock = struct { + reg: Reg, + src_sel: Field, + clk_en: Field, +}; + +inline fn timerClock(comptime g: Group, comptime t: Timer) TimerClock { + return switch (g) { + .timg0 => switch (t) { + .t0 => .{ + .reg = peri_clk_ctrl20, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_CLK_EN_V), + }, + .t1 => .{ + .reg = peri_clk_ctrl20, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_CLK_EN_V), + }, + }, + .timg1 => switch (t) { + .t0 => .{ + .reg = peri_clk_ctrl21, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_CLK_EN_V), + }, + .t1 => .{ + .reg = peri_clk_ctrl21, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_CLK_EN_V), + }, + }, + }; +} + +/// Select a timer's function clock. Comptime instance because the field pairing really does differ +/// per (group, timer) - four different bit positions in two registers. +pub fn setClockSource(comptime g: Group, comptime t: Timer, src: ClockSource) void { + const c = comptime timerClock(g, t); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + c.reg.modify(.{c.src_sel.is(@intFromEnum(src))}); +} + +/// The timer's function-clock gate, distinct from the group's bus clock in `clkrst`. Defaults to 1 +/// at power-on (hp_sys_clkrst_reg.h: REG_TIMERGRP0_T0_CLK_EN default 1). +pub fn setClockEnabled(comptime g: Group, comptime t: Timer, on: bool) void { + const c = comptime timerClock(g, t); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + c.reg.modify(.{c.clk_en.is(@intFromBool(on))}); +} + +// ------------------------------------------------------------------------ general purpose timer + +pub const Direction = enum { up, down }; + +/// Prescaler on the function clock. 2 is the smallest the hardware accepts and 65536 the largest, +/// encoded as 0 (timer_ll.h:191-199). The divider counter is reset in a second store afterwards, +/// exactly as IDF does it: without that the new divider only takes effect after the old one's +/// current period ends, so the first tick after a change is the wrong length. +pub fn setDivider(g: Group, t: Timer, div: u32) void { + std.debug.assert(div >= 2 and div <= 65536); + const cfg = tReg(a_config, g, t); + cfg.modify(.{divider.is(if (div >= 65536) 0 else div)}); + cfg.modify(.{divcnt_rst.is(1)}); +} + +pub fn setDirection(g: Group, t: Timer, dir: Direction) void { + tReg(a_config, g, t).modify(.{increase.is(@intFromBool(dir == .up))}); +} + +/// Reload the counter from `TxLOADLO`/`TxLOADHI` automatically on every alarm. +pub fn setAutoReload(g: Group, t: Timer, on: bool) void { + tReg(a_config, g, t).modify(.{autoreload.is(@intFromBool(on))}); +} + +pub fn setCounterEnabled(g: Group, t: Timer, on: bool) void { + tReg(a_config, g, t).modify(.{counter_en.is(@intFromBool(on))}); +} + +pub fn setAlarmEnabled(g: Group, t: Timer, on: bool) void { + tReg(a_config, g, t).modify(.{alarm_en.is(@intFromBool(on))}); +} + +/// The 54-bit alarm value. Low word first would be equally correct - the comparator only sees the +/// pair - but IDF writes high then low (timer_ll.h:279-283) and matching its order keeps the write +/// trace comparable. +pub fn setAlarmValue(g: Group, t: Timer, value: u64) void { + tReg(a_alarm_hi, g, t).modify(.{alarm_value_hi.is(@truncate(value >> 32))}); + tReg(a_alarm_lo, g, t).writeRaw(@truncate(value)); +} + +/// The value a reload puts into the counter, whether triggered by `load` or by an auto-reload. +pub fn setLoadValue(g: Group, t: Timer, value: u64) void { + tReg(a_load_hi, g, t).modify(.{load_value_hi.is(@truncate(value >> 32))}); + tReg(a_load_lo, g, t).writeRaw(@truncate(value)); +} + +pub fn getLoadValue(g: Group, t: Timer) u64 { + const hi: u64 = tReg(a_load_hi, g, t).get(load_value_hi); + return (hi << 32) | tReg(a_load_lo, g, t).raw(); +} + +/// Copy the load value into the counter now. `TIMG_TxLOAD_REG` is a whole-word write-to-trigger +/// register: the value written is irrelevant, so this is a bare store rather than a field write. +pub fn load(g: Group, t: Timer) void { + tReg(a_load, g, t).writeRaw(1); +} + +/// The counter, through the capture handshake described at the top of this file. +/// +/// Returns null rather than spinning forever if the peripheral never acknowledges: with the group's +/// bus clock gated off, or the timer's function clock gated off, `UPDATE` never clears, and hanging +/// inside a HAL call with no output is the worst possible way to report that. The bound is the same +/// 10,000 reads `systimer.read` uses. +pub fn read(g: Group, t: Timer) ?u64 { + const upd = tReg(a_update, g, t); + + // Ask for a capture. IDF assigns to the struct bitfield, which is a read-modify-write of a word + // whose only other bits are reserved, so `modify` is both the honest operation and the one that + // produces the same store. + upd.modify(.{update.is(1)}); + + var spins: u32 = 0; + while (upd.get(update) != 0) { + spins += 1; + if (spins > 10_000) return null; + } + + const hi: u64 = tReg(a_hi, g, t).get(count_hi); + return (hi << 32) | tReg(a_lo, g, t).raw(); +} + +// ----------------------------------------------------------------------------------- watchdog + +const a_wdtconfig0 = mmio.addr(regs.TIMG_WDTCONFIG0_REG(0)); +const a_wdtconfig1 = mmio.addr(regs.TIMG_WDTCONFIG1_REG(0)); +const a_wdtconfig2 = mmio.addr(regs.TIMG_WDTCONFIG2_REG(0)); +const a_wdtconfig3 = mmio.addr(regs.TIMG_WDTCONFIG3_REG(0)); +const a_wdtconfig4 = mmio.addr(regs.TIMG_WDTCONFIG4_REG(0)); +const a_wdtconfig5 = mmio.addr(regs.TIMG_WDTCONFIG5_REG(0)); +const a_wdtfeed = mmio.addr(regs.TIMG_WDTFEED_REG(0)); +const a_wdtwprotect = mmio.addr(regs.TIMG_WDTWPROTECT_REG(0)); + +const wdt_en = Field.of(regs.TIMG_WDT_EN_S, regs.TIMG_WDT_EN_V); +const wdt_conf_update_en = Field.of(regs.TIMG_WDT_CONF_UPDATE_EN_S, regs.TIMG_WDT_CONF_UPDATE_EN_V); +const wdt_flashboot_mod_en = Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V); +const wdt_cpu_reset_length = Field.of(regs.TIMG_WDT_CPU_RESET_LENGTH_S, regs.TIMG_WDT_CPU_RESET_LENGTH_V); +const wdt_sys_reset_length = Field.of(regs.TIMG_WDT_SYS_RESET_LENGTH_S, regs.TIMG_WDT_SYS_RESET_LENGTH_V); +const wdt_clk_prescale = Field.of(regs.TIMG_WDT_CLK_PRESCALE_S, regs.TIMG_WDT_CLK_PRESCALE_V); +const wdt_divcnt_rst = Field.of(regs.TIMG_WDT_DIVCNT_RST_S, regs.TIMG_WDT_DIVCNT_RST_V); + +/// The write-protect key, and also `TIMG_WDT_WKEY`'s reset value: protection is on whenever the +/// register holds anything *else* (timer_group_reg.h, TIMG_WDT_WKEY, default 1356348065). IDF's +/// `mwdt_ll_write_protect_disable` writes this exact constant (mwdt_ll.h:243). +pub const wkey: u32 = 0x50D8_3AA1; + +/// What IDF writes to re-enable protection (mwdt_ll.h:233). Any non-key value would do; using the +/// same one keeps the register comparable against IDF's. +const wkey_locked: u32 = 0; + +// The headers carry no reset-value macro to check `wkey` against - `TIMG_WDT_WKEY_V` is the field +// mask, 0xffffffff - so the constant is copied from the two places that state it: the register +// description's "default: 1356348065" and mwdt_ll.h:243's 0x50D83AA1. The unit test at the end of +// this file pins those two against each other, which is the only check available without a chip. + +/// MWDT stages, each with its own timeout and its own action. Stage 0 fires first; a stage that is +/// not fed escalates to the next. +pub const Stage = enum(u2) { stage0 = 0, stage1 = 1, stage2 = 2, stage3 = 3 }; + +/// What a stage does when it expires (mwdt_ll.h:23-26). +pub const Action = enum(u2) { + off = 0, + interrupt = 1, + reset_cpu = 2, + reset_system = 3, +}; + +/// Length of the reset pulse a `reset_cpu`/`reset_system` stage asserts (mwdt_ll.h:28-35). +pub const ResetLength = enum(u3) { + ns_100 = 0, + ns_200 = 1, + ns_300 = 2, + ns_400 = 3, + ns_500 = 4, + ns_800 = 5, + us_1_6 = 6, + us_3_2 = 7, +}; + +/// A group's MWDT with write protection lifted, and the only way to reach an MWDT operation: +/// +/// const wdt = timg.unlock(.timg1); +/// defer wdt.release(); +/// wdt.setStage(.stage0, ticks, .reset_system); +/// +/// The handle exists because a watchdog register write made while protection is on is silently +/// dropped - no fault, no status bit, just a watchdog that keeps its old timeout - and that is not a +/// mistake worth making twice. +pub const Watchdog = struct { + group: Group, + + /// Re-enable write protection. Not idempotent-with-`unlock` in the composable sense that + /// `clkrst.Guard` is: the hardware has one key register and no nesting count, so an inner + /// `release` really does lock an outer caller out. There is nothing in this HAL that nests. + pub inline fn release(self: Watchdog) void { + gReg(a_wdtwprotect, self.group).writeRaw(wkey_locked); + } + + /// WDTCONFIG0-5 are shadowed; the hardware only takes them at a `CONF_UPDATE_EN` pulse + /// (mwdt_ll.h:121-122). Write-to-trigger, so this is a single deliberate store. + inline fn commit(self: Watchdog) void { + gReg(a_wdtconfig0, self.group).modify(.{wdt_conf_update_en.is(1)}); + } + + inline fn config0(self: Watchdog) Reg { + return gReg(a_wdtconfig0, self.group); + } + + /// The stage's action bits and its timeout live in different registers - the action in + /// WDTCONFIG0, the timeout in WDTCONFIG2+stage - which is why this takes both at once + /// (mwdt_ll.h:98-123). `timeout` is in MWDT clock cycles, i.e. after the prescaler. + pub fn setStage(self: Watchdog, stage: Stage, timeout: u32, action: Action) void { + self.config0().modify(.{stageAction(stage).is(@intFromEnum(action))}); + gReg(stageHoldAddr(stage), self.group).writeRaw(timeout); + self.commit(); + } + + /// Turn one stage off without disturbing its timeout (mwdt_ll.h:131-152). + pub fn disableStage(self: Watchdog, stage: Stage) void { + self.config0().modify(.{stageAction(stage).is(@intFromEnum(Action.off))}); + self.commit(); + } + + pub fn getStageTimeout(self: Watchdog, stage: Stage) u32 { + return gReg(stageHoldAddr(stage), self.group).raw(); + } + + /// Prescaler from the MWDT's source clock (XTAL on this chip - mwdt_ll.h:273-283 asserts it and + /// selects nothing). 1 to 65535; IDF's default is 20000, which gives 500 ticks/us + /// (mwdt_ll.h:20). + pub fn setPrescaler(self: Watchdog, prescaler: u32) void { + std.debug.assert(prescaler >= 1 and prescaler <= 0xffff); + gReg(a_wdtconfig1, self.group).modify(.{wdt_clk_prescale.is(prescaler)}); + self.commit(); + } + + pub fn setCpuResetLength(self: Watchdog, len: ResetLength) void { + self.config0().modify(.{wdt_cpu_reset_length.is(@intFromEnum(len))}); + self.commit(); + } + + pub fn setSysResetLength(self: Watchdog, len: ResetLength) void { + self.config0().modify(.{wdt_sys_reset_length.is(@intFromEnum(len))}); + self.commit(); + } + + /// Flash-boot protection: a second, independent way for this watchdog to run. It ignores + /// `WDT_EN` entirely (mwdt_ll.h:186-188), it defaults to 1, and a group reset re-arms it - so + /// clearing it is part of every sane bring-up, and `clkrst.resetPeripheral` does it. + pub fn setFlashbootEnabled(self: Watchdog, on: bool) void { + self.config0().modify(.{wdt_flashboot_mod_en.is(@intFromBool(on))}); + self.commit(); + } + + /// Start or stop the watchdog. No `CONF_UPDATE_EN` pulse: `mwdt_ll_enable` and `_disable` + /// (mwdt_ll.h:61-77) do not, so neither does this. Disabling does *not* stop flash-boot mode. + pub fn setEnabled(self: Watchdog, on: bool) void { + self.config0().modify(.{wdt_en.is(@intFromBool(on))}); + } + + pub fn isEnabled(self: Watchdog) bool { + return self.config0().get(wdt_en) == 1; + } + + /// Reset the count and the stage. `TIMG_WDTFEED_REG` is a whole-word write-to-trigger register, + /// so the value is irrelevant (mwdt_ll.h:219-222). + pub fn feed(self: Watchdog) void { + gReg(a_wdtfeed, self.group).writeRaw(1); + } + + /// Reset the watchdog's clock divider counter. Write-to-trigger, in WDTCONFIG1 alongside the + /// prescaler. + pub fn resetDividerCount(self: Watchdog) void { + gReg(a_wdtconfig1, self.group).modify(.{wdt_divcnt_rst.is(1)}); + self.commit(); + } +}; + +/// Lift write protection and hand back the only handle that can touch the MWDT. +pub fn unlock(g: Group) Watchdog { + gReg(a_wdtwprotect, g).writeRaw(wkey); + return .{ .group = g }; +} + +/// Feed a watchdog, protection dance included. The one MWDT operation that is worth a shortcut, +/// because it is the one called from a loop. +pub fn feed(g: Group) void { + const wdt = unlock(g); + defer wdt.release(); + wdt.feed(); +} + +/// True if write protection is currently on, i.e. the key register holds something other than the +/// key. Reads the register, so it reports the hardware rather than what this module last wrote. +pub fn isWriteProtected(g: Group) bool { + return gReg(a_wdtwprotect, g).raw() != wkey; +} + +inline fn stageAction(stage: Stage) Field { + // Stage 0 is at the *top* of the word (bits 30:29) and stage 3 at 24:23, i.e. the stages run + // downwards through the register. The four are a uniform 2 bits apart, but in the reverse of + // the obvious direction, so they are looked up rather than computed. + return switch (stage) { + .stage0 => Field.of(regs.TIMG_WDT_STG0_S, regs.TIMG_WDT_STG0_V), + .stage1 => Field.of(regs.TIMG_WDT_STG1_S, regs.TIMG_WDT_STG1_V), + .stage2 => Field.of(regs.TIMG_WDT_STG2_S, regs.TIMG_WDT_STG2_V), + .stage3 => Field.of(regs.TIMG_WDT_STG3_S, regs.TIMG_WDT_STG3_V), + }; +} + +inline fn stageHoldAddr(stage: Stage) u32 { + // WDTCONFIG2 holds stage 0's timeout and WDTCONFIG5 stage 3's; the mapping is off by two and + // there is no macro that says so, so it comes from mwdt_ll.h:100-116. + return switch (stage) { + .stage0 => a_wdtconfig2, + .stage1 => a_wdtconfig3, + .stage2 => a_wdtconfig4, + .stage3 => a_wdtconfig5, + }; +} + +test "the two strides are the documented ones" { + // 0x1000 between groups (timer_group_reg.h:14, REG_TIMG_BASE) and 0x24 between the two timers + // of a group. Both are asserted against the macros at comptime above; this pins the numbers so + // a header change shows up as a failing test with a value in it, not only as a compile error. + try std.testing.expectEqual(@as(u32, 0x1000), group_stride); + try std.testing.expectEqual(@as(u32, 0x24), timer_stride); +} + +test "the write-protect key is the register's reset value" { + try std.testing.expectEqual(@as(u32, 1_356_348_065), wkey); +} diff --git a/src/hal/uart.zig b/src/hal/uart.zig new file mode 100644 index 0000000..7c891a2 --- /dev/null +++ b/src/hal/uart.zig @@ -0,0 +1,622 @@ +//! The HP UART controllers: UART0-4. +//! +//! Out of scope on purpose: UHCI/DMA, RS485, IrDA, hardware and software flow control, the wakeup +//! machinery, and LP_UART (which is a different block behind a different clock tree, not an +//! instance of this one). +//! +//! Three things about this peripheral cost real debugging time, and all three are structural rather +//! than incidental: +//! +//! **Half the configuration registers are shadowed.** The registers whose macro name ends `_SYNC` - +//! UART_CLKDIV_SYNC, UART_CONF0_SYNC, and a dozen more - are not the live configuration. A write +//! lands in a shadow that the core clock domain ignores until UART_REG_UPDATE is set, at which +//! point the hardware copies the shadow across and clears the bit itself. Reads come back from the +//! shadow, so a read-modify-write composes correctly and a read-back proves nothing about what the +//! transmitter is currently using. Every mutator here therefore ends in `update()`, which is +//! exactly what ESP-IDF does: `uart_ll_update` (uart_ll.h:85-89) sets the bit and spins on it, and +//! every `*_sync` writer in that file calls it (set_stop_bits at uart_ll.h:793, set_parity at 828, +//! set_data_bit_num at 1023, set_loop_back at 1439, the FIFO resets at 735 and 750). Omitting it +//! does not fail loudly: the register reads back as asked and the wire keeps the old setting. +//! +//! **Reading offset 0x000 pops the RX FIFO.** `UART_FIFO_REG`'s only field is annotated `RO` in +//! uart_reg.h:18 and that annotation is wrong in the way that matters - the read is the pop. A +//! generic "snapshot the block" loop therefore eats received bytes, which is why the differential +//! harness carries a per-peripheral deny-list of offsets. Writes to the same address push a byte, +//! and must be full 32-bit stores: a byte store on this bus is a read-modify-write, so it would pop +//! a byte in order to push one (uart_ll.h:716-724 says so and is the reason `pushByte` uses +//! `writeRaw`). +//! +//! **UART0 is the console.** Resetting it clears UART_CLKDIV, the console turns to garbage +//! mid-sentence and the board dies on a watchdog reset with nothing readable to explain it. That was +//! measured on this board. Nothing here resets UART0 implicitly, `reset()` refuses instance 0, and +//! the differential suite uses UART1. +//! +//! The clock path is two dividers in series and they live in different blocks: HP_SYS_CLKRST holds +//! the integer pre-divider (`REG_UARTn_SCLK_DIV_NUM`) and the source select, the UART itself holds +//! the 12.4 fixed-point divider. `setBaudrate` drives both, because neither alone spans the range. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// UART0-4. LP_UART (ESP-IDF's port 5) is a separate peripheral and not modelled here. +pub const count = 5; + +/// SOC_UART_FIFO_LEN, soc_caps.h:655. Both directions; the TX count register reports how many bytes +/// are queued, so free space is this minus that. +pub const fifo_len = 128; + +// ------------------------------------------------------------------------------ register blocks +// One 0x1000-byte block per instance (soc.h:20, `REG_UART_BASE(i) = DR_REG_UART_BASE + i*0x1000`). +// Every register is reached through a RegArray so the stride is checked against the headers rather +// than assumed, and a wrong instance index is a bounds assert rather than a write into UART2. + +fn regArray(comptime offset: u32) type { + return mmio.RegArray( + regs.DR_REG_UART0_BASE + offset, + regs.DR_REG_UART0_BASE + 0x1000 + offset, + count, + ); +} + +const fifo = regArray(0x00); +const clkdiv_sync = regArray(0x14); +const status = regArray(0x1c); +const conf0_sync = regArray(0x20); +const clk_conf = regArray(0x88); +const reg_update = regArray(0x98); + +// CLKDIV_SYNC: a 12.4 fixed-point divider, with the fraction not adjacent to the integer part. +const clkdiv = Field.of(regs.UART_CLKDIV_S, regs.UART_CLKDIV_V); +const clkdiv_frag = Field.of(regs.UART_CLKDIV_FRAG_S, regs.UART_CLKDIV_FRAG_V); + +// CONF0_SYNC: the data format, the FIFO resets and the loopback switch all share this word, which is +// why every one of them is a read-modify-write and not a `write`. +const parity = Field.of(regs.UART_PARITY_S, regs.UART_PARITY_V); +const parity_en = Field.of(regs.UART_PARITY_EN_S, regs.UART_PARITY_EN_V); +const bit_num = Field.of(regs.UART_BIT_NUM_S, regs.UART_BIT_NUM_V); +const stop_bit_num = Field.of(regs.UART_STOP_BIT_NUM_S, regs.UART_STOP_BIT_NUM_V); +const loopback = Field.of(regs.UART_LOOPBACK_S, regs.UART_LOOPBACK_V); +const rxfifo_rst = Field.of(regs.UART_RXFIFO_RST_S, regs.UART_RXFIFO_RST_V); +const txfifo_rst = Field.of(regs.UART_TXFIFO_RST_S, regs.UART_TXFIFO_RST_V); + +// STATUS: live counters, so read-only and never worth comparing between two runs. +const rxfifo_cnt = Field.of(regs.UART_RXFIFO_CNT_S, regs.UART_RXFIFO_CNT_V); +const txfifo_cnt = Field.of(regs.UART_TXFIFO_CNT_S, regs.UART_TXFIFO_CNT_V); + +const tx_sclk_en = Field.of(regs.UART_TX_SCLK_EN_S, regs.UART_TX_SCLK_EN_V); +const rx_sclk_en = Field.of(regs.UART_RX_SCLK_EN_S, regs.UART_RX_SCLK_EN_V); + +/// UART_REG_UPDATE, the commit bit for the whole `_SYNC` family. `R/W/SC`: the hardware clears it +/// when the copy is done. +const reg_update_bit = Field.of(regs.UART_REG_UPDATE_S, regs.UART_REG_UPDATE_V); + +// -------------------------------------------------------------------------------- clock control +// The source select and the integer pre-divider are one register apart, and not in the register the +// names suggest: for UARTn the select is in PERI_CLK_CTRL(110+n) and the pre-divider is in +// PERI_CLK_CTRL(111+n). That is not a typo in this file - uart_ll.h:463-475 writes +// `peri_clk_ctrl110.reg_uart0_clk_src_sel` while uart_ll.h:558-568 writes +// `peri_clk_ctrl111.reg_uart0_sclk_div_num`, so ctrl111 holds UART0's divider *and* UART1's select. + +const peri_clk_ctrl = mmio.RegArray( + regs.HP_SYS_CLKRST_PERI_CLK_CTRL110_REG, + regs.HP_SYS_CLKRST_PERI_CLK_CTRL111_REG, + 6, // ctrl110..ctrl115: five selects and five dividers, overlapping by one +); + +// All five instances place these fields at the same shifts in their respective registers +// (hp_sys_clkrst_reg.h: every REG_UARTn_CLK_SRC_SEL_S is 24, every REG_UARTn_SCLK_DIV_NUM_S is 0, +// every REG_UARTn_CLK_EN_S is 26), so one macro triple each describes all of them. +const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_V); +const sclk_div_num = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_V); +const sclk_en = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_V); + +/// The three clock sources an HP UART can take, with the encoding from uart_ll.h:447-461. +pub const ClockSource = enum(u2) { + /// The 40 MHz crystal. The only source whose frequency is exact, which is why it is the default + /// for anything that has to interoperate. + xtal = 0, + /// RC_FAST, the always-on oscillator. Nominally 20 MHz and uncalibrated - it varies with + /// temperature and part, so a baud rate derived from `nominalHz` here is approximate. + rtc = 1, + /// A fixed 80 MHz tap off the system PLL. Needed for the high rates: the 12-bit integer divider + /// runs out below about 5 kBd from XTAL. + pll_f80m = 2, + + /// The nominal frequency to hand `setBaudrate`. Nominal is exact for `xtal` and `pll_f80m` and a + /// datasheet typical for `rtc`; the real clock tree can be reconfigured, so a caller that has + /// changed it must pass its own number instead. + pub fn nominalHz(self: ClockSource) u32 { + return switch (self) { + .xtal => 40_000_000, + .rtc => 20_000_000, + .pll_f80m => 80_000_000, + }; + } +}; + +pub const WordLength = enum(u2) { + // uart_types.h:57-60. The encoding is (bits - 5), which is why it starts at zero. + bits5 = 0, + bits6 = 1, + bits7 = 2, + bits8 = 3, +}; + +pub const StopBits = enum(u2) { + // uart_types.h:68-70. There is no encoding for zero stop bits, so the enum starts at 1 and 0 is + // reserved by the hardware. + one = 1, + one_and_half = 2, + two = 3, +}; + +pub const Parity = enum(u2) { + // uart_types.h:78-80: bit 1 is "parity enabled", bit 0 is odd/even. `disable` is 0, so the + // odd/even bit is not part of it - see `setParity` for why that matters. + disable = 0, + even = 2, + odd = 3, +}; + +/// One UART instance. A value type holding nothing but the index, so it costs nothing at runtime and +/// every register access folds to a constant address when the index is known. +pub const Uart = struct { + num: u8, + + pub fn init(num: u8) Uart { + std.debug.assert(num < count); + return .{ .num = num }; + } + + // ----------------------------------------------------------------------------- the commit bit + + /// Copy the `_SYNC` shadow registers into the core clock domain and wait for the hardware to + /// acknowledge by clearing the bit (uart_ll.h:85-89). + /// + /// Bounded, where ESP-IDF's `while (hw->reg_update.reg_update);` is not: a UART whose core clock + /// is gated off never clears the bit, and on a board with no debugger an infinite spin is + /// indistinguishable from a crash. 4096 spins is several thousand times the observed cost of a + /// commit, which takes a handful of core-clock cycles. Returns false rather than panicking so a + /// caller can report the peripheral instead of losing the console. + pub fn update(self: Uart) bool { + const r = reg_update.at(self.num); + r.modify(.{reg_update_bit.is(1)}); + return r.waitFor(reg_update_bit, 0, 4096); + } + + // ---------------------------------------------------------------------------- clocks and reset + + /// Reset the block. Refuses UART0. + /// + /// UART0 carries this board's console. A reset clears UART_CLKDIV to its power-on 694, the + /// console's output becomes garbage part-way through whatever it was printing, and the board + /// takes a watchdog reset a moment later - measured, not theorised. There is no "and then put + /// the divider back" version of this that is safe, because the damage is done between the two + /// stores. + pub fn reset(self: Uart) void { + std.debug.assert(self.num != 0); + switch (self.num) { + 1 => clkrst.resetPeripheral(.uart1), + 2 => clkrst.resetPeripheral(.uart2), + 3 => clkrst.resetPeripheral(.uart3), + 4 => clkrst.resetPeripheral(.uart4), + else => unreachable, + } + } + + /// The core (baud-generating) clock, as distinct from the APB bus clock that + /// `clkrst.setClockEnabled` handles. Both are needed: the bus clock makes the registers + /// answer, this one makes the shift registers move - and `update()` is one of the things that + /// stops working without it. + /// + /// Two gates in two blocks, per uart_ll.h:379-397: HP_SYS_CLKRST's per-instance `CLK_EN`, which + /// sits in the *select* register PERI_CLK_CTRL(110+n) and not the divider one next to it, and + /// the UART's own TX and RX enables in UART_CLK_CONF. Interrupts are masked over the first + /// because PERI_CLK_CTRL is shared with unrelated peripherals. + pub fn setCoreClockEnabled(self: Uart, on: bool) void { + const v: u32 = @intFromBool(on); + { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.selectReg().modify(.{sclk_en.is(v)}); + } + clk_conf.at(self.num).modify(.{ tx_sclk_en.is(v), rx_sclk_en.is(v) }); + } + + /// Select the clock the baud generator divides down. Read-modify-write of a register shared with + /// other peripherals, so interrupts are masked (uart_ll.h:477-481 makes the equivalent point by + /// refusing to compile outside `PERIPH_RCC_ATOMIC`). + pub fn setClockSource(self: Uart, src: ClockSource) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.selectReg().modify(.{clk_src_sel.is(@intFromEnum(src))}); + } + + pub fn clockSource(self: Uart) ClockSource { + // Encoding 3 is not defined; IDF's getter (uart_ll.h:509-524) maps `default` to RTC, so + // reporting the same thing keeps a round-trip through both implementations consistent. + return switch (self.selectReg().get(clk_src_sel)) { + 0 => .xtal, + 2 => .pll_f80m, + else => .rtc, + }; + } + + /// PERI_CLK_CTRL(110+n): where this instance's source select and core clock gate live. + inline fn selectReg(self: Uart) Reg { + return peri_clk_ctrl.at(self.num); + } + + /// PERI_CLK_CTRL(111+n): where this instance's integer pre-divider lives. One register above + /// the select, which is the trap this pair of accessors exists to contain. + inline fn dividerReg(self: Uart) Reg { + return peri_clk_ctrl.at(self.num + 1); + } + + // ------------------------------------------------------------------------------------- baud + + /// The two dividers a baud rate decomposes into, computed exactly as + /// `_uart_ll_set_baudrate` (uart_ll.h:532-588) does. + pub const Divider = struct { + /// HP_SYS_CLKRST's integer pre-divider, 1-256. Stored as `sclk - 1` in an 8-bit field. + sclk: u32, + /// The UART's own divider, integer part, 12 bits. + int: u32, + /// The UART's own divider, sixteenths. + frag: u32, + }; + + /// Decompose a baud rate, or fail if the hardware cannot express it. + /// + /// The arithmetic, line by line against uart_ll.h: + /// + /// 541 max_div = UART_CLKDIV_V = 0xfff - the UART divider's integer part is 12 bits + /// 542 sclk = ceil(sclk_freq / (max_div * baud)) the smallest pre-divide that brings + /// the remaining ratio inside 12 bits + /// 545 reject sclk == 0 or sclk > 256 256 = SCLK_DIV_NUM_V + 1 + /// 549 clk_div = (sclk_freq << 4) / (baud * sclk) the ratio in sixteenths + /// 551 int = clk_div >> 4 + /// 552 frag = clk_div & 0xf + /// 555+ the field written is sclk - 1 + /// + /// The `<< 4` is IDF's fixed-point scale, not a fudge: CLKDIV_FRAG is a count of sixteenths of a + /// source-clock period added to every bit time, so `clk_div` is the exact ratio rounded down to + /// 1/16 of a tick. Two deliberate departures from the C, neither of which changes a result: + /// + /// * The `ceil` denominator is 64-bit here as it is there (uart_ll.h:542 casts `max_div` to + /// `uint64_t`), and `sclk_freq << 4` is *also* computed in 64 bits. In C that shift is + /// `uint32_t` and overflows above 268.4 MHz; no P4 UART source is anywhere near that (the + /// fastest is PLL_F80M at 80 MHz), so the two agree on every reachable input while this one + /// has no undefined case. + /// * `baud == 0` returns null rather than false-with-registers-untouched; same outcome, but the + /// caller cannot ignore it by accident. + pub fn divider(baud: u32, sclk_freq: u32) ?Divider { + if (baud == 0) return null; + const max_div: u64 = clkdiv.max(); // UART_CLKDIV_V + const denom = max_div * baud; + const sclk: u64 = (@as(u64, sclk_freq) + denom - 1) / denom; + if (sclk == 0 or sclk > @as(u64, sclk_div_num.max()) + 1) return null; + const clk_div: u64 = (@as(u64, sclk_freq) << 4) / (@as(u64, baud) * sclk); + return .{ + .sclk = @intCast(sclk), + .int = @intCast(clk_div >> 4), + .frag = @intCast(clk_div & 0xf), + }; + } + + /// Program a baud rate. Returns false, having touched nothing, if it is unreachable from this + /// source frequency. + /// + /// Store order follows uart_ll.h:550-576 exactly - integer part, fraction, pre-divider, commit - + /// because the intermediate states are visible to the transmitter of a UART that is already + /// running, and because a write-trace comparison against IDF would otherwise differ on ordering + /// while agreeing on the final registers. The two CLKDIV_SYNC stores are separate for the same + /// reason: IDF's two bitfield assignments are two read-modify-writes of that word. + pub fn setBaudrate(self: Uart, baud: u32, sclk_freq: u32) bool { + const d = divider(baud, sclk_freq) orelse return false; + const div = clkdiv_sync.at(self.num); + div.modify(.{clkdiv.is(d.int)}); + div.modify(.{clkdiv_frag.is(d.frag)}); + { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.dividerReg().modify(.{sclk_div_num.is(d.sclk - 1)}); + } + _ = self.update(); + return true; + } + + /// The baud rate the registers currently describe, by inverting the above + /// (uart_ll.h:590-615). Integer division both ways, so this is not exactly the value passed to + /// `setBaudrate` - it is what the hardware will actually produce, which is the more useful + /// number. + pub fn baudrate(self: Uart, sclk_freq: u32) u32 { + const div = clkdiv_sync.at(self.num).raw(); + const int = (div >> clkdiv.shift) & clkdiv.unshiftedMask(); + const frag = (div >> clkdiv_frag.shift) & clkdiv_frag.unshiftedMask(); + const sclk = self.dividerReg().get(sclk_div_num) + 1; + const ticks = ((@as(u64, int) << 4) | frag) * sclk; + if (ticks == 0) return 0; + return @intCast((@as(u64, sclk_freq) << 4) / ticks); + } + + // ------------------------------------------------------------------------------ data format + + /// uart_ll.h:1020-1024. + pub fn setWordLength(self: Uart, w: WordLength) void { + conf0_sync.at(self.num).modify(.{bit_num.is(@intFromEnum(w))}); + _ = self.update(); + } + + /// uart_ll.h:790-794. + pub fn setStopBits(self: Uart, s: StopBits) void { + conf0_sync.at(self.num).modify(.{stop_bit_num.is(@intFromEnum(s))}); + _ = self.update(); + } + + /// uart_ll.h:817-832. + /// + /// Note what IDF does *not* do: disabling parity leaves UART_PARITY - the odd/even select bit - + /// at whatever it was, because the value 0 for "disabled" carries no odd/even information and + /// writing bit 0 of it would be writing a zero the caller never asked for. So `.disable` clears + /// `parity_en` only. Reproduced here because otherwise a differential run diverges by one bit + /// after any sequence that sets odd parity and then disables it. + pub fn setParity(self: Uart, p: Parity) void { + const c = conf0_sync.at(self.num); + const v = @intFromEnum(p); + if (p != .disable) c.modify(.{parity.is(v & 1)}); + c.modify(.{parity_en.is((v >> 1) & 1)}); + _ = self.update(); + } + + /// All three format fields, in IDF's order. Three commits rather than one, matching what + /// calling IDF's three setters does: the format of a UART mid-transmission is not atomic on + /// this hardware either way, and diverging here would be a difference with no benefit. + pub fn setFormat(self: Uart, w: WordLength, p: Parity, s: StopBits) void { + self.setWordLength(w); + self.setParity(p); + self.setStopBits(s); + } + + pub fn wordLength(self: Uart) WordLength { + return @enumFromInt(conf0_sync.at(self.num).get(bit_num)); + } + + pub fn stopBits(self: Uart) StopBits { + // Encoding 0 is not a legal stop-bit count. The hardware's reset value is 1, and nothing + // here can write 0, so an out-of-range read means the block is unclocked or was reset + // under us - reported as `one` rather than an illegal enum value, which would be UB. + return switch (conf0_sync.at(self.num).get(stop_bit_num)) { + 2 => .one_and_half, + 3 => .two, + else => .one, + }; + } + + /// uart_ll.h:834-841: parity is only meaningful when enabled, so the odd/even bit is not + /// reported unless it is. + pub fn parityMode(self: Uart) Parity { + const c = conf0_sync.at(self.num).raw(); + if ((c >> parity_en.shift) & 1 == 0) return .disable; + return if ((c >> parity.shift) & 1 == 1) .odd else .even; + } + + // ------------------------------------------------------------------------------------- FIFO + + /// Bytes waiting in the RX FIFO (uart_ll.h:763-766). + pub fn rxCount(self: Uart) u32 { + return status.at(self.num).get(rxfifo_cnt); + } + + /// Bytes queued in the TX FIFO. + pub fn txCount(self: Uart) u32 { + return status.at(self.num).get(txfifo_cnt); + } + + /// Free space in the TX FIFO (uart_ll.h:775-780: the total, minus what is queued). + pub fn txFree(self: Uart) u32 { + return fifo_len - self.txCount(); + } + + /// Push one byte. A full 32-bit store, because a narrower one becomes a read-modify-write on + /// this bus and the read would pop a received byte (uart_ll.h:716-724). + pub inline fn pushByte(self: Uart, byte: u8) void { + fifo.at(self.num).writeRaw(byte); + } + + /// Pop one byte. The read *is* the pop - see this file's header on why offset 0x000 is on the + /// differential harness's no-read list. + pub inline fn popByte(self: Uart) u8 { + return @truncate(fifo.at(self.num).raw()); + } + + /// Discard everything received. Assert, commit, deassert, commit: `rxfifo_rst` lives in a + /// shadow register, so without the commits the hardware never sees either edge + /// (uart_ll.h:733-739). + pub fn resetRxFifo(self: Uart) void { + const c = conf0_sync.at(self.num); + c.modify(.{rxfifo_rst.is(1)}); + _ = self.update(); + c.modify(.{rxfifo_rst.is(0)}); + _ = self.update(); + } + + /// uart_ll.h:748-754. Same shape, and the same reason for it. + pub fn resetTxFifo(self: Uart) void { + const c = conf0_sync.at(self.num); + c.modify(.{txfifo_rst.is(1)}); + _ = self.update(); + c.modify(.{txfifo_rst.is(0)}); + _ = self.update(); + } + + // --------------------------------------------------------------------------------- loopback + + /// Tie TX back to RX inside the block (uart_ll.h:1437-1441). The pads are not involved, which + /// makes it the only way to exercise a UART end to end with nothing wired to the board - it is + /// how the FIFO and format paths can be tested at all here. + pub fn setLoopback(self: Uart, on: bool) void { + conf0_sync.at(self.num).modify(.{loopback.is(@intFromBool(on))}); + _ = self.update(); + } + + pub fn loopbackEnabled(self: Uart) bool { + return conf0_sync.at(self.num).get(loopback) == 1; + } + + // ------------------------------------------------------------------------------ pin routing + + /// This instance's TX signal index in the GPIO matrix. The names in IDF's map are + /// `UARTn_TXD_PAD_OUT_IDX` (gpio_sig_map.h:28-52) and they are consecutive in steps of three, + /// but the step is not relied on: each is named. + pub fn txSignal(self: Uart) u32 { + return switch (self.num) { + 0 => regs.UART0_TXD_PAD_OUT_IDX, + 1 => regs.UART1_TXD_PAD_OUT_IDX, + 2 => regs.UART2_TXD_PAD_OUT_IDX, + 3 => regs.UART3_TXD_PAD_OUT_IDX, + 4 => regs.UART4_TXD_PAD_OUT_IDX, + else => unreachable, + }; + } + + /// This instance's RX signal index. Numerically equal to the TX one - the matrix's input and + /// output signal spaces are separate namespaces that happen to share indices for a duplex + /// peripheral - which is exactly why routing RX with `matrixOut` silently does nothing useful. + pub fn rxSignal(self: Uart) u32 { + return switch (self.num) { + 0 => regs.UART0_RXD_PAD_IN_IDX, + 1 => regs.UART1_RXD_PAD_IN_IDX, + 2 => regs.UART2_RXD_PAD_IN_IDX, + 3 => regs.UART3_RXD_PAD_IN_IDX, + 4 => regs.UART4_RXD_PAD_IN_IDX, + else => unreachable, + }; + } + + /// Route TX to a pad through the GPIO matrix. + pub fn routeTx(self: Uart, pin: u8) void { + gpio.matrixOut(pin, self.txSignal()); + } + + /// Route a pad to RX through the GPIO matrix, and enable that pad's input buffer - without + /// which the routed signal reads as a constant and the UART receives nothing, which is the + /// single most common way this goes wrong. + pub fn routeRx(self: Uart, pin: u8) void { + gpio.setInputEnable(pin, true); + gpio.matrixIn(pin, self.rxSignal()); + } + + // --------------------------------------------------------------------------------- transfers + + /// Send every byte, blocking until each fits. Bounded only by the FIFO draining, which always + /// progresses while the core clock is on - so unlike a blocking *read* this cannot wait on an + /// event that may never happen. + pub fn write(self: Uart, bytes: []const u8) void { + for (bytes) |b| { + while (self.txFree() == 0) {} + self.pushByte(b); + } + } + + /// Drain up to `buf.len` received bytes and report how many there were. Does not block. + /// + /// Deliberately not blocking: nothing on the other end of a UART is obliged to send, so a + /// blocking read is an unbounded wait, and there is no timer in this HAL's dependency set to + /// bound it with. A caller that wants to wait writes the loop, and owns the decision about what + /// to do when the bytes never come. + pub fn read(self: Uart, buf: []u8) usize { + var n: usize = 0; + const available = self.rxCount(); + while (n < buf.len and n < available) : (n += 1) buf[n] = self.popByte(); + return n; + } + + /// Whether the transmitter has finished: nothing queued in the FIFO. + /// + /// Not the same as "the last bit is on the wire" - the shift register still holds up to one + /// character after the FIFO empties. UART_FSM_STATUS reports that, and this HAL does not model + /// it, so a caller about to cut the clock or reconfigure the format must allow for one more + /// character time. + pub fn txIdle(self: Uart) bool { + return self.txCount() == 0; + } +}; + +// ------------------------------------------------------------------------------------ host tests +// The divider arithmetic is the only part of this file that can be checked without the chip, and it +// is the part most worth checking: every value below is IDF's formula evaluated by hand, so a +// transcription error in `divider` fails here rather than as a garbled console. + +test "40 MHz XTAL, 115200 Bd: one source tick, 12.4 divider does the work" { + // 40e6/(4095*115200) = 0.085 -> ceil = 1. clk_div = (40e6<<4)/115200 = 5555 (5555.55 floored). + // 5555 = 347*16 + 3. + const d = Uart.divider(115200, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 1), d.sclk); + try std.testing.expectEqual(@as(u32, 347), d.int); + try std.testing.expectEqual(@as(u32, 3), d.frag); + // 347 + 3/16 = 347.1875 ticks per bit -> 115,213 Bd in real arithmetic, and 115,211 as the + // hardware's own truncating inverse reports it (see the round-trip test): 0.01% fast either way. +} + +test "80 MHz PLL, 115200 Bd: the fraction differs from the XTAL case, which is the point of it" { + // 80e6/(4095*115200) = 0.17 -> 1. clk_div = (80e6<<4)/115200 = 11111 = 694*16 + 7. + const d = Uart.divider(115200, 80_000_000).?; + try std.testing.expectEqual(@as(u32, 1), d.sclk); + try std.testing.expectEqual(@as(u32, 694), d.int); + try std.testing.expectEqual(@as(u32, 7), d.frag); +} + +test "a rate low enough to need the pre-divider" { + // 300 Bd from 40 MHz: 40e6/300 = 133,333 ticks per bit, far past 12 bits. + // ceil(40e6/(4095*300)) = ceil(32.6) = 33. clk_div = (40e6<<4)/(300*33) = 64,646 = 4040*16 + 6. + const d = Uart.divider(300, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 33), d.sclk); + try std.testing.expectEqual(@as(u32, 4040), d.int); + try std.testing.expectEqual(@as(u32, 6), d.frag); + try std.testing.expect(d.int <= 0xfff); + try std.testing.expect(d.sclk <= 256); +} + +test "unreachable rates are rejected rather than rounded" { + // Zero is IDF's explicit early return (uart_ll.h:538). + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(0, 40_000_000)); + // 10 Bd from 40 MHz needs a pre-divide of ceil(40e6/40950) = 977, past the 8-bit field's 256. + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(10, 40_000_000)); +} + +test "the sclk == 0 rejection is unreachable except from a zero source frequency" { + // Worth pinning down, because the obvious reading of uart_ll.h:545 is wrong. `sclk` is a + // *ceiling*, so for any non-zero source frequency it is at least 1 - asking for 4 MBd from a + // 1 kHz clock does NOT fail here, it yields sclk = 1 and a divider of zero, and IDF programs + // that just as happily. The only input that trips the branch is sclk_freq == 0. + const absurd = Uart.divider(4_000_000, 1000).?; + try std.testing.expectEqual(@as(u32, 1), absurd.sclk); + try std.testing.expectEqual(@as(u32, 0), absurd.int); + try std.testing.expectEqual(@as(u32, 0), absurd.frag); + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(115200, 0)); +} + +test "the pre-divider field stores sclk - 1, so the reachable rates stop at 256 ticks" { + // The boundary IDF checks at uart_ll.h:545: sclk may be 256 because the field holds sclk-1. + // From 40 MHz the last rate inside it is 39 Bd, at a pre-divide of 251; 38 Bd needs 257. + const ok = Uart.divider(39, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 251), ok.sclk); + try std.testing.expectEqual(@as(u32, 4086), ok.int); + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(38, 40_000_000)); +} + +test "the divider round-trips through the baud rate the hardware will really produce" { + // What `baudrate()` computes, without a chip: the inverse of the same arithmetic. + const d = Uart.divider(115200, 40_000_000).?; + const ticks = ((@as(u64, d.int) << 4) | d.frag) * d.sclk; + const actual: u32 = @intCast((@as(u64, 40_000_000) << 4) / ticks); + // 347 + 3/16 = 347.1875 ticks per bit, and 40e6*16/5555 truncates to 115,211 Bd: 0.01% fast. + try std.testing.expectEqual(@as(u32, 115_211), actual); +} diff --git a/src/io/chip.zig b/src/io/chip.zig new file mode 100644 index 0000000..c5e4127 --- /dev/null +++ b/src/io/chip.zig @@ -0,0 +1,100 @@ +//! The ESP32-P4 half of the scheduler's machine seam: time, critical sections, idling, entropy and +//! a way to print when nothing else works. +//! +//! This is the only file in `src/io/` that touches the chip, and it is deliberately thin - eight +//! functions - because everything above it is portable and gets tested on the host through +//! `host.zig`, which implements the same eight. Nothing here re-derives a register address or a +//! clock: the timebase is `hal.systimer`, the critical section is `hal.intr`'s (which is +//! `clkrst.Guard` under another name, so a critical section written against either module is the +//! same one), and the console is the mask-ROM `ets_printf` that `soc.rom` already declares. + +const std = @import("std"); +const hal = @import("hal"); +const soc = @import("soc"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +/// The one timebase on this board that does not move. SYSTIMER is XTAL/2.5 = 16 MHz, fixed +/// (`clk_tree_defs.h:196-198`, and `hal/systimer.zig:28-31`): it is not derived from the CPU clock, +/// which the bootloader left at a measured ~90 MHz and which nothing here reconfigures. Using the +/// cycle counter instead would make every timeout in the stack wrong by a factor of four the moment +/// somebody raises the PLL. +pub const ticks_hz: u64 = hal.systimer.hz; + +/// Last value the counter gave us, so `ticks` can be monotonic even when the read fails. +var last_ticks: u64 = 0; + +/// Bring the timebase up. Idempotent, and specifically does *not* reprogram the clock source or +/// divider - `hal.systimer.init` documents why: re-running that on a live counter makes the +/// timebase jump, which would corrupt every deadline already computed from it. +pub fn init() void { + hal.systimer.init(); + last_ticks = hal.systimer.read(.unit0) orelse 0; +} + +/// The 52-bit counter, through the update/valid handshake `hal.systimer.read` implements. +/// +/// A gated-off systimer never sets VALUE_VALID, and `hal.systimer.read` reports that as `null` +/// rather than hanging. Returning the previous value there is the only safe answer: returning zero +/// would send time backwards, and every deadline in the scheduler is an unsigned comparison against +/// it, so one backwards step would turn every pending sleep into "already expired". +pub fn ticks() u64 { + const t = hal.systimer.read(.unit0) orelse return last_ticks; + last_ticks = t; + return t; +} + +/// A critical section against interrupt handlers. `hal.intr.Guard` nests correctly - `release` only +/// sets mstatus.MIE if MIE was set on entry - so the scheduler can take one inside a handler. +pub const Guard = hal.intr.Guard; + +pub inline fn mask() Guard { + return hal.intr.mask(); +} + +/// Wait for something to change. Called with interrupts in whatever state the caller had them, +/// which is normally enabled, and free to return at any time: every caller re-checks its condition. +/// +/// This **spins** rather than issuing `wfi`, and that is a decision worth stating. `wfi` is what a +/// power-managed system would do, but it can only be woken by an interrupt, and on this board there +/// is no timer interrupt to wake it: `hal/systimer.zig` exposes the counters and none of the six +/// comparators, and nothing in `hal.intr` is wired to them. A `wfi` with a pending deadline and no +/// alarm configured is a hang, and a `wfi` with no deadline at all is a hang that the scheduler's +/// deadlock watchdog cannot even report, because the watchdog needs to keep running to fire. So +/// this spins on the counter, which costs power and finds every bug. +/// +/// The follow-up is small and worth doing when power matters: a SYSTIMER comparator (`TARGET0`, +/// `SYSTIMER_TARGET0_INT`) routed through `hal.intr` would let this be `wfi` with an exact wake. +pub fn idle(deadline: ?u64) void { + _ = deadline; + // One counter read is ~20 cycles of handshake, which is a fine spin quantum and re-reads the + // register the caller is about to compare against anyway. + _ = ticks(); +} + +/// Print, when the machinery that would normally print has failed. `ets_printf` is a mask ROM +/// address (`esp32p4.rom.ld:24`, re-declared by this project's linker script), so it allocates +/// nothing, takes no lock, and works before or after any of this project's code is functional. +pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void { + soc.rom.print(fmt, args); +} + +/// The hardware random number register: `WDEV_RND_REG`, which on a pre-v3 P4 die is +/// `LP_SYSTEM_REG_RNG_DATA_REG` (`components/soc/esp32p4/register/hw_ver1/soc/wdev_reg.h:16`) at +/// `DR_REG_LP_SYS_BASE + 0x1a4` = 0x501101a4. `esp_random` reads exactly this register and nothing +/// else (`components/esp_hw_support/hw_random.c:78,86`). +/// +/// How much entropy is behind it is a separate question, and the answer for this image is "not +/// established" - see `p4.zig`'s `random` for what is done about that. The register itself is real. +const rng_data = mmio.Reg.at(regs.LP_SYSTEM_REG_RNG_DATA_REG); + +pub inline fn entropyWord() u32 { + return rng_data.raw(); +} + +/// Anything else the machine can contribute to a seed. The cycle counter is not a second timebase - +/// it is the same instant measured with a different, unknown divisor - but its low bits carry the +/// jitter of however many bus stalls happened since reset, which is exactly what a seed wants. +pub inline fn noise() u64 { + return soc.cycles(); +} diff --git a/src/io/context.zig b/src/io/context.zig new file mode 100644 index 0000000..eee8d26 --- /dev/null +++ b/src/io/context.zig @@ -0,0 +1,212 @@ +//! The machine half of the scheduler: what a suspended task is, and how to hand it the CPU. +//! +//! A suspended task is **three words**: `sp`, `fp`, `pc`. Nothing else is stored, and that is not a +//! shortcut - it is the same trick `std.Io.fiber` uses for aarch64, riscv64 and x86_64 +//! (`/usr/lib/zig/std/Io/fiber.zig:7-24`), and it is worth understanding before reading the asm, +//! because "the context switch saves twelve callee-saved registers" is the shape everybody expects +//! and it is not what happens here. +//! +//! The switch declares **every register it does not save as clobbered**. The register allocator +//! then spills whatever is live across the switch onto the switching task's own stack, as part of +//! the calling function's frame, and reloads it when that frame is resumed. So the callee-saved +//! registers *are* saved - by the compiler, into the stack the task already owns, and only the ones +//! that actually hold something. Compiled for this target with `-OReleaseSmall` the spill set +//! around one switch is `ra, s1-s11, fs0-fs11` (verified by reading `-femit-asm` output for +//! riscv32 with `+f`), i.e. 24 words, and it shrinks to nothing in a leaf that holds no live state. +//! A hand-written switch that stores all 24 unconditionally would be both bigger and slower. +//! +//! The exact register set, for the ESP32-P4's rv32imafc: +//! +//! * **saved by this file, in `Context`:** `sp` (x2), `fp` (x8), and the resume address. +//! * **saved by the compiler, because they are clobbered:** `ra` (x1), `t0-t2` (x5-x7), +//! `s1` (x9), `a0` and `a2-a7` (x10, x12-x17), `s2-s11` (x18-x27), `t3-t6` (x28-x31), all 32 +//! `f` registers, and `fflags`/`frm`. `a1` (x11) is the asm's own in/out operand. +//! * **not switched at all, deliberately:** `x0` is hardwired zero. `gp` (x3) is the linker's +//! global pointer - one value for the whole image, never written by a task. `tp` (x4) is the +//! thread pointer; this image has no thread-locals and one hart, so there is nothing per-task +//! to point at. Listing either as a clobber would ask the register allocator to spill a +//! register it can never reload. +//! +//! **First entry differs from a resume in exactly one way: the resume address.** A resume jumps to +//! the label inside the asm block, lands back inside `contextSwitch`, and returns to a frame whose +//! spills are all where the compiler left them. A first entry jumps to a function's first +//! instruction on a stack that contains nothing at all - so on first entry: +//! +//! * `sp` must satisfy the ABI's entry condition (16-byte aligned on RISC-V; 16-byte aligned +//! *minus one word* on x86-64, where a `call` would have pushed a return address); +//! * `fp` is zero, which terminates a frame-pointer walk rather than following garbage; +//! * `ra` is **garbage**, because there is nowhere to return to. The entry function is therefore +//! `noreturn`: it must end in another context switch, and the compiler emits no `ret` for it. +//! On x86-64, where the return address is a stack slot rather than a register, the slot is +//! filled with `returnTrap` so that a mistake is a named panic instead of a wild jump. +//! +//! Nothing is passed to the entry function in a register. There is no way to: on the resumed side +//! the only defined register is the asm's in/out operand, which holds the *switching* task's +//! `Switch` pointer. `p4.zig` reads the current task from the scheduler instead, which it has +//! already set before switching. + +const std = @import("std"); +const builtin = @import("builtin"); + +/// The ESP32-P4 itself. `std.Io.fiber` covers aarch64, riscv64 and x86_64 but not riscv32 +/// (`fiber.zig:1-4`), so the chip's switch is written here and the host's is borrowed from std - +/// which is what lets the scheduler's tests run on the host at all. +pub const rv32 = builtin.cpu.arch == .riscv32; + +pub const supported = rv32 or std.Io.fiber.supported; + +/// A suspended task's whole machine state. 12 bytes on rv32, 24 on the 64-bit hosts. +pub const Context = if (rv32) extern struct { + sp: usize, + fp: usize, + pc: usize, +} else std.Io.fiber.Context; + +/// Layout-identical to `std.Io.fiber.Switch`; a distinct type only so the riscv32 path does not +/// have to name a std type that does not describe it. +pub const Switch = extern struct { old: *Context, new: *Context }; + +/// What the ABI requires of `sp` at a function's first instruction, on every target here. +pub const stack_align = 16; + +/// Save the current machine state into `s.old` and resume the state in `s.new`. +/// +/// Returns when something switches back to `s.old`. Never returns if nothing does - which is the +/// scheduler's problem, not this function's. +pub inline fn contextSwitch(s: *const Switch) void { + if (rv32) { + _ = asm volatile ( + // a1 holds `s` on the way in, and on the way out holds the `Switch` of whoever resumed us. + // The loads must all happen before `sp` moves: after `lw sp, 0(a2)` this frame is gone. + \\ lw a0, 0(a1) + \\ lw a2, 4(a1) + \\ lla a3, 0f + \\ sw sp, 0(a0) + \\ sw fp, 4(a0) + \\ sw a3, 8(a0) + \\ lw sp, 0(a2) + \\ lw fp, 4(a2) + \\ lw a3, 8(a2) + \\ jr a3 + \\0: + : [received] "={a1}" (-> *const Switch), + : [send] "{a1}" (s), + : .{ + .x1 = true, + .x5 = true, + .x6 = true, + .x7 = true, + .x9 = true, + .x10 = true, + .x12 = true, + .x13 = true, + .x14 = true, + .x15 = true, + .x16 = true, + .x17 = true, + .x18 = true, + .x19 = true, + .x20 = true, + .x21 = true, + .x22 = true, + .x23 = true, + .x24 = true, + .x25 = true, + .x26 = true, + .x27 = true, + .x28 = true, + .x29 = true, + .x30 = true, + .x31 = true, + .f0 = true, + .f1 = true, + .f2 = true, + .f3 = true, + .f4 = true, + .f5 = true, + .f6 = true, + .f7 = true, + .f8 = true, + .f9 = true, + .f10 = true, + .f11 = true, + .f12 = true, + .f13 = true, + .f14 = true, + .f15 = true, + .f16 = true, + .f17 = true, + .f18 = true, + .f19 = true, + .f20 = true, + .f21 = true, + .f22 = true, + .f23 = true, + .f24 = true, + .f25 = true, + .f26 = true, + .f27 = true, + .f28 = true, + .f29 = true, + .f30 = true, + .f31 = true, + .fflags = true, + .frm = true, + .memory = true, + }); + } else { + _ = std.Io.fiber.contextSwitch(@ptrCast(s)); + } +} + +/// The `Context` for a task that has never run, which will enter `entry` on the stack ending at +/// `stack_top`. `stack_top` is exclusive, i.e. one past the last writable byte. +/// +/// Writes to the top of the stack on x86-64 (see the file comment); reads nothing. +pub fn initial(stack_top: usize, entry: *const fn () callconv(.c) noreturn) Context { + const top = stack_top & ~@as(usize, stack_align - 1); + return switch (builtin.cpu.arch) { + .x86_64 => sp: { + // System V x86-64 guarantees `rsp % 16 == 0` *before* the `call` that enters a + // function, so a callee's first instruction sees `rsp % 16 == 8` with the return + // address at `[rsp]`. We arrive by `jmp`, so both halves have to be faked, or the + // first 16-byte-aligned spill in the entry function faults. + const sp = top - 8; + @as(*usize, @ptrFromInt(sp)).* = @intFromPtr(&returnTrap); + break :sp .{ .rsp = sp, .rbp = 0, .rip = @intFromPtr(entry) }; + }, + else => .{ .sp = top, .fp = 0, .pc = @intFromPtr(entry) }, + }; +} + +/// How much of the top of a stack `initial` consumes before the entry function's first frame. +pub const initial_stack_overhead: usize = switch (builtin.cpu.arch) { + .x86_64 => 8, + else => 0, +}; + +fn returnTrap() callconv(.c) noreturn { + @panic("io.p4: a task returned from its entry function; the entry function must be noreturn"); +} + +test "initial context points at the entry function on an aligned stack" { + if (!supported) return error.SkipZigTest; + var stack: [256]u8 align(stack_align) = undefined; + const c = initial(@intFromPtr(&stack) + stack.len, returnTrap); + const sp = switch (builtin.cpu.arch) { + .x86_64 => c.rsp, + else => c.sp, + }; + const pc = switch (builtin.cpu.arch) { + .x86_64 => c.rip, + else => c.pc, + }; + try std.testing.expectEqual(@intFromPtr(&returnTrap), pc); + try std.testing.expect(sp <= @intFromPtr(&stack) + stack.len); + try std.testing.expect(sp > @intFromPtr(&stack)); + // The entry condition the ABI states, per target. + switch (builtin.cpu.arch) { + .x86_64 => try std.testing.expectEqual(@as(usize, 8), sp % stack_align), + else => try std.testing.expectEqual(@as(usize, 0), sp % stack_align), + } +} diff --git a/src/io/host.zig b/src/io/host.zig new file mode 100644 index 0000000..7239f1a --- /dev/null +++ b/src/io/host.zig @@ -0,0 +1,76 @@ +//! The host half of the scheduler's machine seam: the same eight declarations as `chip.zig`, for a +//! machine that is not the chip. +//! +//! This exists so the scheduler itself can be tested where it can be debugged. Everything above +//! `context.zig` and this file is portable - the run queue, the futex, cancellation, deadline +//! arithmetic - and all of it is the part where a mistake is a silent hang on the die. +//! +//! The clock here is **virtual**, not the wall clock, and that is the point. `ticks` returns a +//! counter that only advances when the scheduler idles, so a test that sleeps three tasks for +//! 5 ms, 1 ms and 3 ms finishes instantly and in a fixed order, with no tolerance windows and no +//! flakiness under load. It also means the scheduler's own deadlock watchdog is testable: the +//! no-deadline idle advances the clock too, so a watchdog measured in ticks still fires. + +const std = @import("std"); + +/// The chip's rate, kept identically here so the host exercises the same tick/nanosecond arithmetic +/// (`* 125 / 2`) rather than a rounder number that would hide a division bug. +pub const ticks_hz: u64 = 16_000_000; + +var virtual: u64 = 0; + +pub fn init() void { + virtual = 0; +} + +pub fn ticks() u64 { + return virtual; +} + +/// No interrupts to mask on the host, and the scheduler is single-threaded, so this is a shape +/// rather than a mechanism. It still has to exist: it is what makes the chip's masked regions +/// reachable in a host test. +pub const Guard = struct { + pub inline fn release(_: Guard) void {} +}; + +pub inline fn mask() Guard { + return .{}; +} + +/// Advance the virtual clock. With a deadline, jump straight to it - nothing else can happen in +/// between on a host with no interrupts, so waiting is pure delay. Without one, advance by a +/// millisecond so that a caller spinning on "nothing runnable, nothing scheduled" reaches its +/// watchdog instead of looping forever. +pub fn idle(deadline: ?u64) void { + if (deadline) |d| { + if (d > virtual) virtual = d; + } else { + virtual += ticks_hz / 1000; + } +} + +/// Enough of `ets_printf`'s shape to typecheck the arguments, printed in Zig's own spelling. The +/// crash path is the only caller, and on the host the interesting question is whether it runs at +/// all, not how it looks. +pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void { + std.debug.print("{s} <- {any}\n", .{ std.mem.span(fmt), args }); +} + +/// No hardware RNG. Returns zero rather than something plausible: `p4.zig` mixes this with the +/// clock and its own state, and a zero here makes it obvious in a test that the hardware +/// contribution is absent instead of quietly supplying the entropy the chip is being asked about. +pub inline fn entropyWord() u32 { + return 0; +} + +/// Deliberately constant: a host test that asserts something about the generator's output wants a +/// fixed seed, and the chip's own seed noise is `soc.cycles()`, which no host has. +pub inline fn noise() u64 { + return 0x0420_0cafe_0420; +} + +/// Test-only: move the virtual clock forward by hand. +pub fn advance(n: u64) void { + virtual += n; +} diff --git a/src/io/p4.zig b/src/io/p4.zig new file mode 100644 index 0000000..ef82e02 --- /dev/null +++ b/src/io/p4.zig @@ -0,0 +1,1475 @@ +//! `std.Io` for the ESP32-P4: a cooperative scheduler over caller-provided static stacks. +//! +//! This is the "FreeRTOS for Zig" layer, except that it is not a framework and there is nothing to +//! port to: `std.Io` is an interface, and filling in enough of its vtable buys the whole ecosystem +//! that is written against it. Implementing `futexWait`/`futexWake` and `async`/`await` here is +//! what makes `std.Io.Mutex`, `std.Io.Condition`, `std.Io.Semaphore`, `std.Io.RwLock`, +//! `std.Io.Queue(T)` and `std.Io.Future(T)` work on this chip - none of which appear in this file, +//! because they are built on those primitives in std and need nothing from us +//! (`/usr/lib/zig/std/Io.zig:1587` Mutex, `:1653` Condition, `:1874` Queue). +//! +//! ## What is real and what is not +//! +//! `std.Io.VTable` has 109 entries. Sixteen are implemented here; the other 93 come from +//! `unimplemented.zig`, where each one panics with its own name. That is the deliberate shape of +//! this file: a partial implementation whose gaps announce themselves, rather than a plausible +//! stub that returns zero. Implemented: +//! +//! now clockResolution sleep - the timebase, from hal.systimer +//! async concurrent await cancel - cooperative tasks on static stacks +//! futexWait futexWaitUncancelable futexWake +//! checkCancel recancel swapCancelProtection +//! crashHandler random randomSecure - randomSecure reports EntropyUnavailable; see `random` +//! +//! Not implemented, and therefore a named panic: every `dir*`, `file*`, `net*`, `process*` and +//! `child*` entry, `operate`, `batchAwait*`/`batchCancel`, the four `group*` entries, +//! `lockStderr`/`tryLockStderr`/`unlockStderr`, and `progressParentFile`. So `std.Io.Group`, +//! `std.Io.Select`, and any Reader or Writer that reaches the OS are out; the concurrency and +//! synchronisation types listed above are in. +//! +//! ## The model +//! +//! One hart, no preemption, no timer interrupt. A task runs until it calls something that blocks - +//! `sleep`, `futexWait`, `await`, `cancel`, or `yield` - and the scheduler is entered from inside +//! that call. There is no scheduler task and no scheduler stack: the context that called +//! `Runtime.init` is itself a task (the "main" slot, id 0), so a switch is always task-to-task and +//! the first blocking call in `main` is what starts everything else. `io.async` only *assigns* a +//! slot and marks it ready; the body first executes when the spawning context blocks or yields. +//! That is within the contract - `async` promises a unit of concurrency, not that the work started +//! (`Io.zig:54-58`) - and it is why a `main` that spawns seven tasks and then returns runs none of +//! them. +//! +//! ## Why a futex reduces to "park this task" here, and where that stops being true +//! +//! A futex exists to close a race: between "I checked the value" and "I went to sleep", a waker can +//! change the value and wake nothing, and the waiter sleeps forever. Linux closes it by doing the +//! compare and the enqueue inside the kernel, atomically. +//! +//! On this chip the whole window is closed by the machine. One hart and no preemption means no +//! other *task* can run between the compare and the enqueue - there is no instant at which control +//! could pass to a waker - so a plain compare followed by a plain enqueue is already atomic with +//! respect to every task. The only thing that can intervene is an **interrupt handler**, and +//! ESP-Hosted does post from one (`_h_post_semaphore_from_isr`), so the compare-and-enqueue runs +//! inside `hal.intr.mask()` - two CSR instructions - and `futexWake` runs inside the same critical +//! section. That makes the pair atomic against handlers too, and `futexWake` safe to call from a +//! CLIC handler. +//! +//! This stops being sound the moment either assumption goes: +//! +//! * **A second core.** The P4 has two HP cores and an LP core. A task on core 1 could observe +//! the value, be preempted by nothing at all, and still race a waker on core 0, because +//! masking interrupts on one hart says nothing about the other. A cross-core version needs a +//! real wait queue per address plus an inter-core interrupt, not this. +//! * **Preemption.** Add a timer interrupt that switches tasks and the "no other task can run" +//! argument is gone; the mask covers it, but only because the mask is what the preemption +//! would arrive through. Any preemption source that is not maskable here breaks it. +//! +//! Two further honest limits: waiters are found by scanning the task array rather than by a +//! per-address wait queue (correct, O(tasks), and tasks are counted in single digits), and the scan +//! is in slot order, so `futexWake(ptr, 1)` picks the lowest-numbered waiter rather than the +//! longest-waiting one. Under a contended `std.Io.Mutex` that is a fairness wart, not a +//! correctness one - the run queue itself is FIFO - but a heavily contended lock could starve a +//! high-numbered slot. +//! +//! ## Static footprint +//! +//! Nothing here allocates, and nothing here is `comptime`-sized by a count. Measured on a real +//! riscv32 build of this file (`footprint` below asserts both columns): +//! +//! @sizeOf(Task) 80 bytes per slot, in .bss - 12 of them are the suspended machine state +//! @sizeOf(Runtime) 152 bytes including the embedded main task +//! stack caller's whatever the slot declares +//! +//! On top of a slot's stack, and only while a slot is running an `io.async`/`io.concurrent` body, +//! sit two copies carved off the top of that same stack: the argument tuple and the result. Both +//! are a few tens of bytes for a typical call, and `claim` refuses a slot whose stack cannot hold +//! them and still leave `min_stack_bytes`, in which case `async` runs the work eagerly instead. +//! +//! So a 7-task ESP-Hosted configuration with 5 KiB stacks is 35 KiB of stack + 560 bytes of slots +//! + 152 bytes of runtime = 35.7 KiB, against the ~128 KiB of L2MEM. +//! +//! Code, from `nm` on a riscv32 `-OReleaseSmall` build of this file against the real `hal`, `soc`, +//! `regs` and `mmio` modules: 2,402 bytes of out-of-line `.text` in 21 symbols for the scheduler +//! and the sixteen implemented entries (`reschedule`, `claim`, `joinChild` and `futexBlock` are +//! private and get inlined into their callers, so a caller pays a little more), plus 1,860 bytes +//! of `.text.unlikely` and 3,722 bytes of `.rodata` strings for the 93 panicking stubs. The stub +//! strings are the single largest line item here, and they are the price of a gap that announces +//! itself: they only survive if the app's root declares a panic handler that prints the message, +//! which `src/main.zig` does. +//! +//! One number worth having when sizing a stack: a task suspended inside the switch is holding +//! about 160 bytes of its own stack for the spills the switch's clobber list forces, on top of +//! whatever its own call chain is using (read off the `addi sp, sp, 160` frame of `reschedule` in +//! the riscv32 asm). `stackUsed` reports the real high-water mark of any slot at any time, from +//! the paint `init` lays down, and `dump` prints it for every task. +//! +//! ## Integrating this +//! +//! Module imports: `hal` (systimer, and `intr.mask` for the critical section), `soc` (`rom.print` +//! and `cycles`), and `regs` + `mmio` (the one RNG register). All four already exist in build.zig. +//! +//! One requirement that is std's rather than this file's: the app's **root module** must declare +//! +//! pub const std_options: std.Options = .{ .page_size_min = 4096 }; +//! +//! Five of the vtable's entries name `Io.File.MemoryMap`, whose `memory` field is +//! `[]align(std.heap.page_size_min) u8` (`/usr/lib/zig/std/Io/File/MemoryMap.zig:18`), and +//! riscv32-freestanding has no default for that (`/usr/lib/zig/std/heap.zig:48`). Defining a +//! function whose return type is `File.MemoryMap.CreateError!File.MemoryMap` is what forces the +//! struct to be laid out, so no stub body can dodge it, and any `std.Io` implementation on this +//! target hits the same wall. It has to be the *root* module, because `std.options` is read as +//! `@import("root").std_options` (`/usr/lib/zig/std/std.zig:112`) - putting the declaration in +//! this file has no effect, which is measured rather than assumed. The value is arbitrary: +//! nothing in this image pages, and it is only ever read for that one field's alignment. +//! +//! ## What is missing that a caller will notice +//! +//! `idle` spins instead of sleeping, because there is no timer interrupt to wake a `wfi` (see +//! `chip.zig`). There is no priority: the run queue is FIFO. And nothing here bounds how long a +//! task may hold the CPU, so one task that never blocks starves the rest - which is the deal with +//! cooperative scheduling, and the reason `sleep` yields even when the deadline has already passed. + +const std = @import("std"); +const builtin = @import("builtin"); +const Io = std.Io; +const Alignment = std.mem.Alignment; +const assert = std.debug.assert; + +const context = @import("context.zig"); +const unimplemented = @import("unimplemented.zig"); + +/// The chip, or the host that tests the chip's scheduler. Both sides implement the same eight +/// declarations; see `chip.zig` for what each one costs on the die. +const machine = if (builtin.os.tag == .freestanding) @import("chip.zig") else @import("host.zig"); + +const Context = context.Context; +const Switch = context.Switch; + +/// Written into every byte of a task's stack at `init`, so that `stackUsed` can report a real +/// high-water mark and `checkStack` can catch an overflow at the next switch instead of at the next +/// mystery. 0xC5 rather than 0 because a zeroed stack is indistinguishable from `.bss`. +const paint: u8 = 0xC5; + +/// The least a slot's stack may be for `async` to use it, over and above the argument and result +/// copies. Below this the slot is skipped and the work runs eagerly on the caller's stack, which is +/// always legal (`Io.zig:54-56`) and better than a stack overflow. +pub const min_stack_bytes: usize = 512; + +/// The runtime that owns the currently running task. A cooperative scheduler needs exactly one +/// global: a task's entry function is jumped to with no arguments - the machine has nowhere to put +/// them, see `context.zig` - so it has to find itself somehow, and this is the pointer it finds +/// itself through. FreeRTOS spells the same thing `pxCurrentTCB`. +/// +/// One `Runtime` per hart, therefore. Installing a second one while the first has live tasks is a +/// programming error, and `init` asserts against it. +var installed: ?*Runtime = null; + +pub const State = enum(u8) { + /// The slot holds no task. `async` may claim it. + free, + /// Runnable, and on the run queue. + ready, + /// Has the CPU. + running, + /// Waiting for a futex, a deadline, or a child task. + blocked, + /// Ran to completion; the result is in `result`, waiting for `await` or `cancel` to collect it. + done, +}; + +/// One unit of concurrency: the stack a task runs on, plus what the scheduler needs to know about +/// it. The caller owns `stack` and nothing else here. +pub const Task = struct { + /// The task's stack. Caller-provided and caller-sized, `.bss` or a static array; the scheduler + /// aligns the top down to the ABI's 16 bytes itself, so any byte slice will do. + /// + /// The main task's stack is empty: it runs on whatever stack `_start` established, and the + /// scheduler never needs to know where that is. + stack: []u8, + + // -------- scheduler-private from here down -------- + + /// `sp`, `fp` and the resume address. Meaningful only while suspended. + ctx: Context = undefined, + state: State = .free, + /// Run queue link. Single-linked FIFO; the queue is only ever walked forwards. + next: ?*Task = null, + /// The address this task is parked on, if it is parked on a futex. + futex: ?*const u32 = null, + /// Machine ticks at which the scheduler must make this task ready again. + deadline: ?u64 = null, + /// The child this task is inside `await`/`cancel` for. + join: ?*Task = null, + /// Who to make ready when this task finishes. + awaiter: ?*Task = null, + /// Whether a cancel request should end the current block. False inside an uncancelable wait. + interruptible: bool = false, + /// A cancel request has been made and not yet delivered. See `Future.cancel`'s doc comment + /// (`Io.zig:1181-1190`): only the *next* cancelation point returns `error.Canceled`. + cancel_pending: bool = false, + /// A cancelation point in this task consumed a request. `await` needs this to decide whether a + /// cancelation it forwarded to a child was actually delivered. + cancel_acked: bool = false, + protection: Io.CancelProtection = .unblocked, + /// The type-erased body, its argument copy and its result storage, as `async` was handed them. + start: *const fn (context: *const anyopaque, result: *anyopaque) void = undefined, + arg: *const anyopaque = undefined, + result: []u8 = &.{}, + /// Slot number, for diagnostics. 0 is the main task. + id: u8 = 0, +}; + +pub const Options = struct { + /// Nanoseconds to add to the systimer to answer `Clock.real`. Zero - the default - means + /// `Clock.real` reads as nanoseconds since the counter started, i.e. since boot: this board has + /// no RTC that survives a reset and no NTP. Set it once real time is known from outside (an + /// SNTP exchange, an HTTP `Date` header) and `Clock.real` becomes real. + real_epoch_offset_ns: i64 = 0, + /// How long the scheduler may find nothing runnable *and* no deadline pending before it + /// declares deadlock and panics with a dump of every task. + /// + /// That state is a deadlock unless an interrupt handler is about to wake somebody, which is a + /// legitimate thing to be waiting for - hence a timeout rather than an immediate panic. Five + /// seconds is long enough for any SDIO transaction on this board and short enough that a lost + /// wakeup is reported rather than looking like a hang. + deadlock_timeout_ms: u32 = 5_000, +}; + +pub const Runtime = struct { + /// Caller-provided slots, one per concurrent task. Index 0 of `tasks()` is `main`, not this. + slots: []Task, + /// The context that called `init`. Never started, only ever resumed. + main: Task, + current: *Task, + ready_head: ?*Task, + ready_tail: ?*Task, + real_epoch_offset_ns: i64, + deadlock_ticks: u64, + rng: Rng, + + /// Take over as this hart's `std.Io`. `slots` must outlive the runtime, and so must `rt` + /// itself: the runtime points at its own `main` field. + /// + /// Brings the timebase up (`hal.systimer.init`, which deliberately does not reprogram the + /// clock source - see `chip.zig`). Touches nothing else on the chip: no interrupt is enabled, + /// no pad is configured, and mstatus.MIE is left exactly as the caller had it. + pub fn init(rt: *Runtime, slots: []Task, opts: Options) void { + comptime { + if (!context.supported) @compileError( + "io/p4.zig: no context switch for this architecture; see context.zig", + ); + } + // `Task.id` is a u8 and 0 is the main task, so 254 slots is the ceiling. Anything close to + // it would have run out of L2MEM for stacks long before. + assert(slots.len < 255); + if (installed) |old| { + if (old != rt) { + var it = old.tasks(); + while (it.next()) |t| assert(t.state == .free or t == &old.main); + } + } + machine.init(); + rt.* = .{ + .slots = slots, + .main = .{ .stack = &.{}, .state = .running, .id = 0 }, + .current = undefined, + .ready_head = null, + .ready_tail = null, + .real_epoch_offset_ns = opts.real_epoch_offset_ns, + .deadlock_ticks = @as(u64, opts.deadlock_timeout_ms) * machine.ticks_hz / 1000, + .rng = .{}, + }; + rt.current = &rt.main; + for (slots, 0..) |*slot, i| { + const stack = slot.stack; + slot.* = .{ .stack = stack, .id = @intCast(i + 1) }; + @memset(stack, paint); + } + installed = rt; + } + + pub fn io(rt: *Runtime) Io { + return .{ .userdata = rt, .vtable = &vtable }; + } + + /// Put the current task at the back of the run queue and let everything else have a turn. + /// + /// Not a `std.Io` entry - there is none - but a cooperative scheduler needs a yield, and code + /// that only has an `Io` can get the same effect from `io.sleep(.zero, .awake)`, which is + /// documented to yield exactly once even for a deadline that has already passed. + pub fn yield(rt: *Runtime) void { + const me = rt.current; + { + const guard = machine.mask(); + defer guard.release(); + me.state = .ready; + rt.push(me); + } + rt.reschedule(); + } + + /// Bytes of `t`'s stack that have ever been written, from the paint laid down by `init`. + /// Zero for the main task, whose stack this scheduler does not own. + pub fn stackUsed(_: *const Runtime, t: *const Task) usize { + var i: usize = 0; + while (i < t.stack.len and t.stack[i] == paint) i += 1; + return t.stack.len - i; + } + + pub fn stackFree(rt: *const Runtime, t: *const Task) usize { + return t.stack.len - rt.stackUsed(t); + } + + /// Every task, main first. + pub fn tasks(rt: *Runtime) Iterator { + return .{ .rt = rt }; + } + + pub const Iterator = struct { + rt: *Runtime, + i: usize = 0, + + pub fn next(it: *Iterator) ?*Task { + const n = it.i; + it.i += 1; + if (n == 0) return &it.rt.main; + if (n - 1 < it.rt.slots.len) return &it.rt.slots[n - 1]; + return null; + } + }; + + /// One line per task, on the ROM console. Allocates nothing and takes no lock, so it is safe + /// from the crash path. + pub fn dump(rt: *Runtime) void { + machine.print("MARK IO_P4 current=%u slots=%u tick=%u\r\n", .{ + @as(u32, rt.current.id), + @as(u32, @intCast(rt.slots.len)), + @as(u32, @truncate(machine.ticks())), + }); + var it = rt.tasks(); + while (it.next()) |t| { + machine.print( + "MARK IO_P4_TASK id=%u state=%u futex=0x%08x deadline=%u join=%u cancel=%u stack=%u/%u\r\n", + .{ + @as(u32, t.id), + @as(u32, @intFromEnum(t.state)), + @as(u32, if (t.futex) |p| @truncate(@intFromPtr(p)) else 0), + @as(u32, @truncate(t.deadline orelse 0)), + @as(u32, if (t.join) |j| j.id else 255), + @as(u32, @intFromBool(t.cancel_pending)), + @as(u32, @intCast(rt.stackUsed(t))), + @as(u32, @intCast(t.stack.len)), + }, + ); + } + } + + // ---------------------------------------------------------------- the run queue + + /// Append to the run queue. Caller holds the interrupt mask. + fn push(rt: *Runtime, t: *Task) void { + t.next = null; + if (rt.ready_tail) |tail| tail.next = t else rt.ready_head = t; + rt.ready_tail = t; + } + + /// Take the front of the run queue. Caller holds the interrupt mask. + fn pop(rt: *Runtime) ?*Task { + const t = rt.ready_head orelse return null; + rt.ready_head = t.next; + if (rt.ready_head == null) rt.ready_tail = null; + t.next = null; + return t; + } + + /// Make a blocked task runnable. Caller holds the interrupt mask. + /// + /// Leaves `futex` and `deadline` set: the waiting task clears its own, after it resumes, and + /// the `.blocked` check here is what stops a second waker counting the same waiter twice. + fn wake(rt: *Runtime, t: *Task) void { + assert(t.state == .blocked); + t.state = .ready; + rt.push(t); + } + + fn free(rt: *Runtime) ?*Task { + for (rt.slots) |*t| if (t.state == .free) return t; + return null; + } + + /// Wake every task whose deadline has arrived. Caller holds the interrupt mask. + fn expire(rt: *Runtime, now_ticks: u64) void { + var it = rt.tasks(); + while (it.next()) |t| { + if (t.state != .blocked) continue; + const d = t.deadline orelse continue; + if (now_ticks >= d) rt.wake(t); + } + } + + /// The soonest deadline anybody is waiting for. Caller holds the interrupt mask. + fn earliest(rt: *Runtime) ?u64 { + var best: ?u64 = null; + var it = rt.tasks(); + while (it.next()) |t| { + if (t.state != .blocked) continue; + const d = t.deadline orelse continue; + if (best == null or d < best.?) best = d; + } + return best; + } + + // ---------------------------------------------------------------- the switch + + /// Give up the CPU. The caller must already have put itself into the state it wants to be found + /// in: `.ready` and on the queue (a yield), or `.blocked` with a wake condition set. + /// + /// Returns when something switches back to this task. `rt.current` is set by whoever switches, + /// before the switch, so on the way back it is already correct. + fn reschedule(rt: *Runtime) void { + const me = rt.current; + assert(me.state != .running); // would be a task that gave up the CPU and stayed runnable + var idle_since: ?u64 = null; + while (true) { + const guard = machine.mask(); + rt.expire(machine.ticks()); + const next = rt.pop(); + if (next) |n| { + if (n == me) { + // An interrupt handler re-queued us before we got here, or we were the only + // runnable task. Either way there is nothing to switch to. + me.state = .running; + guard.release(); + return; + } + rt.current = n; + n.state = .running; + guard.release(); + rt.checkStack(me); + var s: Switch = .{ .old = &me.ctx, .new = &n.ctx }; + context.contextSwitch(&s); + return; + } + const deadline = rt.earliest(); + guard.release(); + rt.waitIdle(deadline, &idle_since); + } + } + + /// Like `reschedule`, for a task that is never coming back. Its `Context` is still needed as + /// somewhere for the switch to write, and is dead the instant the switch completes. + fn retire(rt: *Runtime, me: *Task) noreturn { + var idle_since: ?u64 = null; + while (true) { + const guard = machine.mask(); + rt.expire(machine.ticks()); + const next = rt.pop(); + if (next) |n| { + assert(n != me); // a finished task cannot be runnable + rt.current = n; + n.state = .running; + guard.release(); + var s: Switch = .{ .old = &me.ctx, .new = &n.ctx }; + context.contextSwitch(&s); + @panic("io.p4: a finished task was resumed"); + } + const deadline = rt.earliest(); + guard.release(); + rt.waitIdle(deadline, &idle_since); + } + } + + /// Nothing is runnable. With a deadline that is an ordinary sleep. Without one, the only thing + /// that can save this hart is an interrupt handler calling `futexWake` - a legitimate thing to + /// be waiting for, on a chip whose SDIO slave signals on a GPIO - so give it a bounded chance + /// and then say what happened, loudly, rather than spinning silently forever. + fn waitIdle(rt: *Runtime, deadline: ?u64, idle_since: *?u64) void { + if (deadline == null) { + const now_ticks = machine.ticks(); + if (idle_since.*) |since| { + if (now_ticks -% since > rt.deadlock_ticks) rt.deadlock(); + } else { + idle_since.* = now_ticks; + } + } else { + idle_since.* = null; + } + machine.idle(deadline); + } + + fn checkStack(rt: *Runtime, t: *Task) void { + if (t.stack.len == 0) return; + if (t.stack[0] == paint) return; + machine.print("MARK IO_P4_STACK_OVERFLOW id=%u size=%u\r\n", .{ + @as(u32, t.id), @as(u32, @intCast(t.stack.len)), + }); + rt.dump(); + @panic("io.p4: task stack overflow"); + } + + fn deadlock(rt: *Runtime) noreturn { + machine.print("MARK IO_P4_DEADLOCK no runnable task and no deadline\r\n", .{}); + rt.dump(); + @panic("io.p4: deadlock - every task is blocked and nothing is scheduled to wake them"); + } + + // ---------------------------------------------------------------- time + + fn clockOffset(rt: *const Runtime, clock: Io.Clock) i96 { + return switch (clock) { + .real => rt.real_epoch_offset_ns, + // The systimer's own epoch. `awake` and `boot` are the same thing on a chip that does + // not suspend, and there is no per-task CPU accounting, so the two `cpu_*` clocks + // answer with elapsed time; `clockResolution` reports 0 for them to say so. + .awake, .boot, .cpu_process, .cpu_thread => 0, + }; + } + + /// Absolute machine tick a `Timeout` expires at, or null for `.none`. + fn deadlineOf(rt: *const Runtime, timeout: Io.Timeout) ?u64 { + return switch (timeout) { + .none => null, + .duration => |d| machine.ticks() +| ticksFromNs(d.raw.nanoseconds), + .deadline => |ts| ticksFromNs(ts.raw.nanoseconds - rt.clockOffset(ts.clock)), + }; + } + + // ---------------------------------------------------------------- tasks + + /// Reserve a slot for `start`, with its argument tuple and result storage carved off the top of + /// the slot's own stack. Returns null if there is no free slot, or the slot's stack is too + /// small to hold the copies and still be a stack - in both cases the caller's fallback is to + /// run the work eagerly. + fn claim( + rt: *Runtime, + result_len: usize, + result_alignment: Alignment, + arg: []const u8, + arg_alignment: Alignment, + start: *const fn (context: *const anyopaque, result: *anyopaque) void, + ) ?*Task { + const guard = machine.mask(); + defer guard.release(); + + const t = rt.free() orelse return null; + + const base = @intFromPtr(t.stack.ptr); + const result_at = result_alignment.backward(base + t.stack.len - result_len); + const arg_at = arg_alignment.backward(result_at - arg.len); + if (arg_at < base or arg_at - base < min_stack_bytes + context.initial_stack_overhead) + return null; + + const arg_ptr: [*]u8 = @ptrFromInt(arg_at); + @memcpy(arg_ptr[0..arg.len], arg); + const result_ptr: [*]u8 = @ptrFromInt(result_at); + + t.state = .ready; + t.next = null; + t.futex = null; + t.deadline = null; + t.join = null; + t.awaiter = null; + t.interruptible = false; + t.cancel_pending = false; + t.cancel_acked = false; + t.protection = .unblocked; + t.start = start; + t.arg = @ptrCast(arg_ptr); + t.result = result_ptr[0..result_len]; + t.ctx = context.initial(arg_at, taskEntry); + rt.push(t); + return t; + } + + /// Wait for `child` to finish, collect its result, and give the slot back. + /// + /// `request` is `cancel` rather than `await`. The subtle case is the other one: a cancel + /// request that lands on *us* while we are waiting. `await` has no error to return it through, + /// so the request is forwarded to the child instead, and afterwards re-armed on us if the child + /// never actually observed it - which is what `Threaded.await` does with `recancelInner` + /// (`/usr/lib/zig/std/Io/Threaded.zig:2465-2473`). + fn joinChild(rt: *Runtime, child: *Task, result: []u8, request: bool) void { + const me = rt.current; + assert(child != me); + assert(child.result.len == result.len); + + if (request) rt.requestCancel(child); + var inherited = false; + + while (true) { + if (!request and !inherited and consumeCancel(me)) { + inherited = true; + rt.requestCancel(child); + } + const guard = machine.mask(); + if (child.state == .done) { + guard.release(); + break; + } + child.awaiter = me; + me.join = child; + me.interruptible = !(request or inherited); + me.state = .blocked; + guard.release(); + rt.reschedule(); + me.join = null; + } + + if (result.len != 0) @memcpy(result, child.result); + const acked = child.cancel_acked; + child.awaiter = null; + child.state = .free; + if (inherited and !acked) me.cancel_pending = true; + } + + /// Arm a cancel request on `t`, and end its current wait if that wait is cancelable. + fn requestCancel(rt: *Runtime, t: *Task) void { + const guard = machine.mask(); + defer guard.release(); + switch (t.state) { + .free, .done => return, + .ready, .running, .blocked => {}, + } + t.cancel_pending = true; + if (t.state == .blocked and t.interruptible) rt.wake(t); + } + + /// Park until somebody wakes `ptr`, the deadline arrives, or - if `interruptible` - a cancel + /// request lands. Returns without parking if the value already differs from `expected`, which + /// is a futex's `EAGAIN` and is what keeps a waker that got there first from being lost. + fn futexBlock( + rt: *Runtime, + me: *Task, + ptr: *const u32, + expected: u32, + deadline: ?u64, + interruptible: bool, + ) void { + const guard = machine.mask(); + if (@atomicLoad(u32, ptr, .acquire) != expected) { + guard.release(); + return; + } + if (deadline) |d| if (machine.ticks() >= d) { + guard.release(); + return; + }; + me.futex = ptr; + me.deadline = deadline; + me.interruptible = interruptible; + me.state = .blocked; + guard.release(); + rt.reschedule(); + me.futex = null; + me.deadline = null; + } +}; + +/// Where a task begins. Jumped to, not called: there is no return address and no argument, so the +/// task finds itself through `installed` (see that declaration for why). +fn taskEntry() callconv(.c) noreturn { + const rt = installed.?; + const me = rt.current; + me.start(me.arg, @ptrCast(me.result.ptr)); + + { + const guard = machine.mask(); + defer guard.release(); + me.state = .done; + if (me.awaiter) |a| if (a.state == .blocked and a.join == me) rt.wake(a); + } + rt.checkStack(me); + rt.retire(me); +} + +/// True if `t` has an undelivered cancel request that is not blocked by cancel protection, in which +/// case the request is consumed: only the *next* cancelation point reports it (`Io.zig:1185-1188`). +fn consumeCancel(t: *Task) bool { + if (t.protection == .blocked) return false; + if (!t.cancel_pending) return false; + t.cancel_pending = false; + t.cancel_acked = true; + return true; +} + +fn cast(userdata: ?*anyopaque) *Runtime { + return @ptrCast(@alignCast(userdata)); +} + +// -------------------------------------------------------------------- tick arithmetic + +/// Machine ticks to nanoseconds, exactly. 16 MHz is 62.5 ns per tick, so the conversion is +/// `* 125 / 2` and not a multiply by 62 - which would drift by 0.8%, i.e. 43 seconds a day. +fn nsFromTicks(t: u64) i96 { + return @divTrunc(@as(i96, t) * 125, 2); +} + +/// Nanoseconds to machine ticks, rounded **up**, so a deadline is never reached early. Saturates +/// rather than wrapping or trapping: `Duration.max` is `maxInt(i96)` nanoseconds, which is a +/// request never to wake, and the largest tick count this returns is 36,000 years at 16 MHz. +fn ticksFromNs(ns: i96) u64 { + if (ns <= 0) return 0; + const limit: i96 = @divFloor(@as(i96, std.math.maxInt(u64)) * 125, 2); + if (ns >= limit) return std.math.maxInt(u64); + return @intCast(@divFloor(ns * 2 + 124, 125)); +} + +// -------------------------------------------------------------------- entropy + +/// The random source behind `random`. +/// +/// `WDEV_RND_REG` is real on this die - it is `LP_SYSTEM_REG_RNG_DATA_REG`, and `esp_random` reads +/// exactly it (`chip.zig`) - but the *entropy* behind it is not established for this image, and +/// this file will not claim otherwise. ESP-IDF's own documentation is precise about the conditions +/// (`docs/en/api-reference/system/random.rst:37-46`): the hardware produces true random numbers +/// while the RF subsystem is enabled, or while `bootloader_random_enable` has the SAR ADC entropy +/// source on, or while the IDF second-stage bootloader is running - "if none of the above +/// conditions are true, the output of the RNG should be considered as pseudo-random only". The P4 +/// has no RF at all, this image is not started by the IDF bootloader, and nothing here calls +/// `bootloader_random_enable`. There is a hardware secondary source that is always mixed in - an +/// asynchronous oscillator sampled for metastability (`random.rst:98-101`) - which is why the +/// register is worth reading at all, but "always mixed in" is not the same as "measured on this +/// board", and I cannot measure this board. +/// +/// So: **`random` is not cryptographic**, and says so. It is a PCG32 (`std.Random.Pcg`) seeded from +/// four hardware words mixed with the systimer and the cycle counter, re-stirred with a fresh +/// hardware word whenever the pacing interval has elapsed. `randomSecure` reports +/// `error.EntropyUnavailable` rather than pretending. Closing that gap is a small, separate job: +/// port `bootloader_random_enable`, then measure. +const Rng = struct { + prng: std.Random.Pcg = undefined, + seeded: bool = false, + last_sample: u64 = 0, + + /// ESP-IDF paces `esp_random` at one *byte* per 14 microseconds on this part - + /// `APB_CYCLE_WAIT_NUM` is `CONFIG_ESP_DEFAULT_CPU_FREQ_MHZ * 14` APB cycles per byte + /// (`hw_random.c:41-44`), which at the P4's 1:1 CPU:APB ratio is the ~75 kHz byte rate the + /// comment there says the RNG was tested at. Draining it faster returns the generator's state + /// rather than new entropy. Expressed in systimer ticks because that is the clock this file + /// trusts, and because the constant in IDF assumes a 360 MHz CPU that this board is not + /// running. + const sample_ticks: u64 = machine.ticks_hz * 14 / 1_000_000; + + fn fill(r: *Rng, buffer: []u8) void { + if (!r.seeded) r.seed(); + r.stir(); + r.prng.fill(buffer); + } + + /// Four paced hardware words, which costs ~56 microseconds of spinning, once, on the first call + /// to `random`. Deliberately not done in `init`: a runtime that never asks for randomness + /// should not pay for it, and `init` runs before anything can tolerate a delay. + fn seed(r: *Rng) void { + var s: u64 = machine.noise() ^ (machine.ticks() << 17); + for (0..4) |_| { + s = s *% 0x9E37_79B9_7F4A_7C15 ^ r.sample(); + } + r.prng = .init(s); + r.seeded = true; + } + + /// Mix in a fresh hardware word if one is due. Never waits: the caller asked for random bytes, + /// not for a delay, and the generator is sound without it. + fn stir(r: *Rng) void { + const now_ticks = machine.ticks(); + if (now_ticks -% r.last_sample < sample_ticks) return; + r.last_sample = now_ticks; + r.prng.s ^= machine.entropyWord(); + } + + /// One hardware word, waiting out the pacing interval first. + fn sample(r: *Rng) u32 { + const target = r.last_sample +% sample_ticks; + while (machine.ticks() -% r.last_sample < sample_ticks) machine.idle(target); + r.last_sample = machine.ticks(); + return machine.entropyWord(); + } +}; + +// -------------------------------------------------------------------- the vtable + +/// Start from "nothing is implemented" and overwrite what is. The list of assignments below is the +/// authoritative answer to "what works": anything not named here panics with its own name. +const vtable: Io.VTable = build: { + var v = unimplemented.vtable; + + v.now = now; + v.clockResolution = clockResolution; + v.sleep = sleep; + + v.async = asyncTask; + v.concurrent = concurrent; + v.await = awaitTask; + v.cancel = cancelTask; + + v.futexWait = futexWait; + v.futexWaitUncancelable = futexWaitUncancelable; + v.futexWake = futexWake; + + v.checkCancel = checkCancel; + v.recancel = recancel; + v.swapCancelProtection = swapCancelProtection; + + v.crashHandler = crashHandler; + v.random = random; + v.randomSecure = randomSecure; + + break :build v; +}; + +fn now(userdata: ?*anyopaque, clock: Io.Clock) Io.Timestamp { + const rt = cast(userdata); + return .{ .nanoseconds = nsFromTicks(machine.ticks()) + rt.clockOffset(clock) }; +} + +fn clockResolution(_: ?*anyopaque, clock: Io.Clock) Io.Clock.ResolutionError!Io.Duration { + return switch (clock) { + // 1e9/16e6 = 62.5 ns, truncated: `Duration` counts whole nanoseconds and the tick is not a + // whole number of them. The conversion in `nsFromTicks` keeps the half. + .real, .awake, .boot => .fromNanoseconds(@divTrunc(std.time.ns_per_s, @as(i96, machine.ticks_hz))), + // Zero means "unsupported" (`Io.zig:787-790`). There is one process and no per-task CPU + // accounting, so `now` answers these with elapsed time and this says not to believe it. + .cpu_process, .cpu_thread => .zero, + }; +} + +fn sleep(userdata: ?*anyopaque, timeout: Io.Timeout) Io.Cancelable!void { + // `.none` means no timeout at all, i.e. nothing to wait for. Matches `Threaded.sleep` + // (`Threaded.zig:11577`), and notably is *not* a cancelation point. + if (timeout == .none) return; + + const rt = cast(userdata); + const me = rt.current; + if (consumeCancel(me)) return error.Canceled; + const deadline = rt.deadlineOf(timeout).?; + + while (true) { + { + const guard = machine.mask(); + defer guard.release(); + if (machine.ticks() >= deadline) { + // The deadline has already passed, but a `sleep` that returns without ever leaving + // the CPU makes `io.sleep(.zero, .awake)` useless as a yield - and it is the only + // yield `std.Io` exposes. So go round the run queue exactly once. + me.deadline = null; + me.interruptible = true; + me.state = .ready; + rt.push(me); + } else { + me.deadline = deadline; + me.interruptible = true; + me.state = .blocked; + } + } + rt.reschedule(); + me.deadline = null; + if (consumeCancel(me)) return error.Canceled; + if (machine.ticks() >= deadline) return; + } +} + +fn asyncTask( + userdata: ?*anyopaque, + result: []u8, + result_alignment: Alignment, + arg: []const u8, + arg_alignment: Alignment, + start: *const fn (context: *const anyopaque, result: *anyopaque) void, +) ?*Io.AnyFuture { + const rt = cast(userdata); + const t = rt.claim(result.len, result_alignment, arg, arg_alignment, start) orelse { + // No slot: run it here and now. `await` will be a no-op (`Io.zig:54-56`). + start(arg.ptr, result.ptr); + return null; + }; + return @ptrCast(t); +} + +fn concurrent( + userdata: ?*anyopaque, + result_len: usize, + result_alignment: Alignment, + arg: []const u8, + arg_alignment: Alignment, + start: *const fn (context: *const anyopaque, result: *anyopaque) void, +) Io.ConcurrentError!*Io.AnyFuture { + const rt = cast(userdata); + // `concurrent` promises the caller can block on something the task will unblock. A cooperative + // task can do that - it runs whenever the caller blocks - so the only failure is running out of + // slots. Unlike `async` there is no eager fallback: running the body inline is exactly the + // guarantee `concurrent` exists to rule out. + const t = rt.claim(result_len, result_alignment, arg, arg_alignment, start) orelse + return error.ConcurrencyUnavailable; + return @ptrCast(t); +} + +fn awaitTask(userdata: ?*anyopaque, any_future: *Io.AnyFuture, result: []u8, _: Alignment) void { + const rt = cast(userdata); + rt.joinChild(@ptrCast(@alignCast(any_future)), result, false); +} + +fn cancelTask(userdata: ?*anyopaque, any_future: *Io.AnyFuture, result: []u8, _: Alignment) void { + const rt = cast(userdata); + rt.joinChild(@ptrCast(@alignCast(any_future)), result, true); +} + +fn futexWait(userdata: ?*anyopaque, ptr: *const u32, expected: u32, timeout: Io.Timeout) Io.Cancelable!void { + const rt = cast(userdata); + const me = rt.current; + // Cancelation is checked before the value, which is the order `Threaded` gets from calling + // `Syscall.start()` before the futex syscall (`Threaded.zig:986`): an already-canceled task + // does not acquire a lock on its way out. + if (consumeCancel(me)) return error.Canceled; + rt.futexBlock(me, ptr, expected, rt.deadlineOf(timeout), true); + if (consumeCancel(me)) return error.Canceled; +} + +fn futexWaitUncancelable(userdata: ?*anyopaque, ptr: *const u32, expected: u32) void { + const rt = cast(userdata); + rt.futexBlock(rt.current, ptr, expected, null, false); +} + +fn futexWake(userdata: ?*anyopaque, ptr: *const u32, max_waiters: u32) void { + const rt = cast(userdata); + const guard = machine.mask(); + defer guard.release(); + var woken: u32 = 0; + var it = rt.tasks(); + while (it.next()) |t| { + if (woken >= max_waiters) break; + if (t.state != .blocked) continue; + if (t.futex != ptr) continue; + rt.wake(t); + woken += 1; + } +} + +fn checkCancel(userdata: ?*anyopaque) Io.Cancelable!void { + if (consumeCancel(cast(userdata).current)) return error.Canceled; +} + +fn recancel(userdata: ?*anyopaque) void { + const me = cast(userdata).current; + assert(!me.cancel_pending); // `recancel` with a request already pending + me.cancel_pending = true; +} + +fn swapCancelProtection(userdata: ?*anyopaque, new: Io.CancelProtection) Io.CancelProtection { + const me = cast(userdata).current; + const old = me.protection; + me.protection = new; + return old; +} + +/// Called from `std.debug`'s panic path (`/usr/lib/zig/std/debug.zig:536`) and from the segfault +/// handler (`:1641`), both of which go on to print the panic themselves. +/// +/// So this prints and **returns**; it does not park. Parking here would swallow the panic message, +/// which is the one thing worth having. No allocation, no lock, no scheduling: `dump` is straight +/// `ets_printf`, and marking the crashing task cancel-protected keeps any cleanup that runs after +/// this from being interrupted by a cancelation - the same thing `Threaded.crashHandler` does +/// (`Threaded.zig:2066-2072`). +fn crashHandler(userdata: ?*anyopaque) void { + const rt = cast(userdata); + rt.current.cancel_pending = false; + rt.current.protection = .blocked; + machine.print("MARK IO_P4_CRASH\r\n", .{}); + rt.dump(); +} + +fn random(userdata: ?*anyopaque, buffer: []u8) void { + cast(userdata).rng.fill(buffer); +} + +/// See `Rng`: the hardware register is real, its entropy on this image is not established, and +/// guessing is worse than saying so. +fn randomSecure(_: ?*anyopaque, _: []u8) Io.RandomSecureError!void { + return error.EntropyUnavailable; +} + +// -------------------------------------------------------------------- static declaration helper + +/// A whole runtime as one `.bss` object: `task_count` stacks of `stack_bytes` each, the slots that +/// describe them, and the `Runtime`. +/// +/// ``` +/// var pool: p4.Static(7, 5 * 1024) = .{}; +/// const io = pool.init(.{}).io(); +/// ``` +pub fn Static(comptime task_count: usize, comptime stack_bytes: usize) type { + comptime { + if (stack_bytes < min_stack_bytes) @compileError("stack_bytes is smaller than min_stack_bytes"); + if (stack_bytes % context.stack_align != 0) @compileError( + "stack_bytes must be a multiple of the ABI's stack alignment, so that every slot's " ++ + "stack top is aligned without waste", + ); + } + return struct { + stacks: [task_count][stack_bytes]u8 align(context.stack_align) = undefined, + slots: [task_count]Task = undefined, + runtime: Runtime = undefined, + + /// Total `.bss` this declaration costs. + pub const size = @sizeOf(@This()); + + pub fn init(self: *@This(), opts: Options) *Runtime { + for (&self.slots, &self.stacks) |*slot, *stack| slot.* = .{ .stack = stack }; + self.runtime.init(&self.slots, opts); + return &self.runtime; + } + }; +} + +// -------------------------------------------------------------------- tests +// +// These run on the host, against `host.zig`'s virtual clock and `std.Io.fiber`'s x86-64 context +// switch, which is the same scheduler with a different two files underneath it. What they cannot +// test is the riscv32 asm in `context.zig` - that only runs on the die. + +const testing = std.testing; + +/// Deliberately file-scope: 8 KiB stacks in a test function's frame is not what a stack is for. +/// Unreferenced outside tests, so it is not analysed - let alone emitted - in a firmware build. +var test_pool: Static(4, 8 * 1024) = .{}; + +fn testRuntime() *Runtime { + return test_pool.init(.{}); +} + +const Trace = struct { + buf: [64]u8 = undefined, + len: usize = 0, + + fn put(t: *Trace, c: u8) void { + t.buf[t.len] = c; + t.len += 1; + } + fn seen(t: *const Trace) []const u8 { + return t.buf[0..t.len]; + } +}; + +/// The numbers quoted in this file's header. Not a tuning knob - a regression alarm: `Task` growing +/// silently is how a 35 KiB budget becomes 40. The chip's figures are the 32-bit column, and they +/// were read off a real riscv32 build; the host runs the 64-bit column so that this test is not +/// vacuous where it can actually run. +pub const footprint = struct { + pub const task_bytes: usize = if (@sizeOf(usize) == 4) 80 else 128; + pub const runtime_bytes: usize = if (@sizeOf(usize) == 4) 152 else 216; + pub const context_bytes: usize = 3 * @sizeOf(usize); +}; + +test footprint { + try testing.expectEqual(footprint.task_bytes, @sizeOf(Task)); + try testing.expectEqual(footprint.runtime_bytes, @sizeOf(Runtime)); + try testing.expectEqual(footprint.context_bytes, @sizeOf(Context)); +} + +fn tick(io: Io, trace: *Trace, mark: u8, rounds: usize) void { + for (0..rounds) |_| { + trace.put(mark); + io.sleep(.zero, .awake) catch return; + } +} + +test "three tasks round-robin through the scheduler" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + var a = io.async(tick, .{ io, &trace, 'a', 3 }); + var b = io.async(tick, .{ io, &trace, 'b', 3 }); + var c = io.async(tick, .{ io, &trace, 'c', 3 }); + a.await(io); + b.await(io); + c.await(io); + + // Spawn order is queue order, and each task yields after every mark, so the interleaving is + // exact. `main` awaits `a` first and so is not in the rotation. + try testing.expectEqualStrings("abcabcabc", trace.seen()); +} + +fn locker(io: Io, m: *Io.Mutex, trace: *Trace, mark: u8) void { + m.lock(io) catch return; + defer m.unlock(io); + trace.put(mark); + // Hold the lock across a yield, so the other task must actually block on the futex rather than + // finding it free. + io.sleep(.zero, .awake) catch {}; + trace.put(std.ascii.toUpper(mark)); +} + +test "std.Io.Mutex serialises two tasks through the futex" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + var m: Io.Mutex = .init; + + var a = io.async(locker, .{ io, &m, &trace, 'a' }); + var b = io.async(locker, .{ io, &m, &trace, 'b' }); + a.await(io); + b.await(io); + + // Interleaved would be "abAB"; serialised is each task's pair adjacent. + try testing.expectEqualStrings("aAbB", trace.seen()); + try testing.expect(m.tryLock()); +} + +fn producer(io: Io, q: *Io.Queue(u32), n: u32) void { + var i: u32 = 0; + while (i < n) : (i += 1) q.putOne(io, i) catch return; + q.close(io); +} + +fn consumer(io: Io, q: *Io.Queue(u32), sum: *u32) void { + while (true) { + const v = q.getOne(io) catch return; + sum.* += v; + } +} + +test "std.Io.Queue passes items between two tasks" { + const rt = testRuntime(); + const io = rt.io(); + // Capacity 2 against 10 items, so both directions block and both directions get woken. + var storage: [2]u32 = undefined; + var q: Io.Queue(u32) = .init(&storage); + var sum: u32 = 0; + + var p = io.async(producer, .{ io, &q, 10 }); + var c = io.async(consumer, .{ io, &q, &sum }); + p.await(io); + c.await(io); + + try testing.expectEqual(@as(u32, 45), sum); +} + +fn napper(io: Io, trace: *Trace, mark: u8, ms: u64) void { + io.sleep(.fromMilliseconds(@intCast(ms)), .awake) catch return; + trace.put(mark); +} + +test "sleep orders three tasks by deadline, not by spawn order" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + const t0 = Io.Timestamp.now(io, .awake); + var a = io.async(napper, .{ io, &trace, 'a', @as(u64, 30) }); + var b = io.async(napper, .{ io, &trace, 'b', @as(u64, 10) }); + var c = io.async(napper, .{ io, &trace, 'c', @as(u64, 20) }); + a.await(io); + b.await(io); + c.await(io); + + try testing.expectEqualStrings("bca", trace.seen()); + // And the clock really moved, by at least the longest sleep. + const elapsed = t0.durationTo(Io.Timestamp.now(io, .awake)); + try testing.expect(elapsed.toMilliseconds() >= 30); +} + +fn napAndCatch(io: Io, trace: *Trace) void { + trace.put('s'); + io.sleep(.fromMilliseconds(1000), .awake) catch |err| switch (err) { + error.Canceled => { + trace.put('x'); + return; + }, + }; + trace.put('e'); // must not be reached +} + +test "a canceled task does not run to completion" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + var f = io.async(napAndCatch, .{ io, &trace }); + // The body has not run yet - `async` only assigned a slot - so let it reach its sleep first. + rt.yield(); + try testing.expectEqualStrings("s", trace.seen()); + + f.cancel(io); + try testing.expectEqualStrings("sx", trace.seen()); +} + +test "cancel before the body ever runs still runs it, and its first cancelation point reports" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + var f = io.async(napAndCatch, .{ io, &trace }); + f.cancel(io); + // Same as `Threaded`: the function always runs; cancelation is delivered at the first + // cancelation point inside it. + try testing.expectEqualStrings("sx", trace.seen()); +} + +fn protected(io: Io, trace: *Trace) void { + const old = io.swapCancelProtection(.blocked); + io.sleep(.fromMilliseconds(5), .awake) catch unreachable; // protection is on: cannot fail + trace.put('p'); + _ = io.swapCancelProtection(old); + io.checkCancel() catch { + trace.put('x'); + return; + }; + trace.put('e'); +} + +test "cancel protection defers delivery to the next unprotected point" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + var f = io.async(protected, .{ io, &trace }); + f.cancel(io); + try testing.expectEqualStrings("px", trace.seen()); +} + +fn waiter(io: Io, word: *std.atomic.Value(u32), trace: *Trace) void { + while (word.load(.acquire) == 0) { + io.futexWait(u32, &word.raw, 0) catch return; + } + trace.put('w'); +} + +fn poster(io: Io, word: *std.atomic.Value(u32), trace: *Trace) void { + trace.put('p'); + word.store(1, .release); + io.futexWake(u32, &word.raw, 1); +} + +test "futexWait parks and futexWake releases exactly one waiter" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + var word: std.atomic.Value(u32) = .init(0); + + var w = io.async(waiter, .{ io, &word, &trace }); + var p = io.async(poster, .{ io, &word, &trace }); + w.await(io); + p.await(io); + + try testing.expectEqualStrings("pw", trace.seen()); +} + +test "futexWait returns immediately when the value already differs" { + const rt = testRuntime(); + const io = rt.io(); + var word: std.atomic.Value(u32) = .init(7); + // Would deadlock if it parked: nobody is going to wake it. + try io.futexWait(u32, &word.raw, 0); +} + +test "futexWaitTimeout unblocks on its deadline with nothing else runnable" { + const rt = testRuntime(); + const io = rt.io(); + var word: std.atomic.Value(u32) = .init(0); + + // Nobody is going to wake this, so the deadline has to - which also drives the scheduler's + // "nothing runnable, one deadline pending" idle path from the main task, the same path the + // deadlock watchdog must *not* fire on. + const t0 = Io.Timestamp.now(io, .awake); + try io.futexWaitTimeout(u32, &word.raw, 0, .{ .duration = .{ + .raw = .fromMilliseconds(25), + .clock = .awake, + } }); + const elapsed = t0.durationTo(Io.Timestamp.now(io, .awake)); + try testing.expect(elapsed.toMilliseconds() >= 25); + try testing.expectEqual(@as(u32, 0), word.load(.monotonic)); +} + +fn add(a: u32, b: u32) u32 { + return a + b; +} + +test "async falls back to running eagerly when every slot is taken" { + const rt = testRuntime(); + const io = rt.io(); + + // Fill all four slots with tasks that will not finish until woken. + var word: std.atomic.Value(u32) = .init(0); + var trace: Trace = .{}; + var held: [4]Io.Future(void) = undefined; + for (&held) |*f| f.* = io.async(waiter, .{ io, &word, &trace }); + + var eager = io.async(add, .{ 20, 22 }); + try testing.expectEqual(@as(?*Io.AnyFuture, null), eager.any_future); + try testing.expectEqual(@as(u32, 42), eager.await(io)); + + word.store(1, .release); + io.futexWake(u32, &word.raw, 4); + for (&held) |*f| f.await(io); + try testing.expectEqualStrings("wwww", trace.seen()); +} + +test "concurrent reports ConcurrencyUnavailable instead of running inline" { + const rt = testRuntime(); + const io = rt.io(); + + var word: std.atomic.Value(u32) = .init(0); + var trace: Trace = .{}; + var held: [4]Io.Future(void) = undefined; + for (&held) |*f| f.* = io.async(waiter, .{ io, &word, &trace }); + + try testing.expectError(error.ConcurrencyUnavailable, io.concurrent(add, .{ 1, 2 })); + + word.store(1, .release); + io.futexWake(u32, &word.raw, 4); + for (&held) |*f| f.await(io); +} + +test "now is monotonic and converts ticks exactly" { + const rt = testRuntime(); + const io = rt.io(); + + const a = Io.Timestamp.now(io, .awake); + try io.sleep(.fromMilliseconds(7), .awake); + const b = Io.Timestamp.now(io, .awake); + try testing.expect(b.nanoseconds >= a.nanoseconds); + try testing.expect(a.durationTo(b).toMilliseconds() >= 7); + + // 62.5 ns a tick, kept exact: one tick is 62 ns and two are 125, not 124. + try testing.expectEqual(@as(i96, 62), nsFromTicks(1)); + try testing.expectEqual(@as(i96, 125), nsFromTicks(2)); + try testing.expectEqual(@as(i96, 1_000_000_000), nsFromTicks(machine.ticks_hz)); + // And back, rounding up so a deadline is never early. + try testing.expectEqual(@as(u64, 1), ticksFromNs(1)); + try testing.expectEqual(@as(u64, 1), ticksFromNs(62)); + try testing.expectEqual(@as(u64, 2), ticksFromNs(63)); + try testing.expectEqual(@as(u64, machine.ticks_hz), ticksFromNs(1_000_000_000)); + try testing.expectEqual(@as(u64, std.math.maxInt(u64)), ticksFromNs(Io.Duration.max.nanoseconds)); + + // `real` is the same counter plus an offset the caller sets; `cpu_*` report resolution 0 to say + // they are not really implemented. + try testing.expectEqual(Io.Duration{ .nanoseconds = 62 }, try io.vtable.clockResolution(io.userdata, .awake)); + try testing.expectEqual(Io.Duration.zero, try io.vtable.clockResolution(io.userdata, .cpu_thread)); + rt.real_epoch_offset_ns = 1_700_000_000 * std.time.ns_per_s; + try testing.expect(Io.Timestamp.now(io, .real).toSeconds() > 1_600_000_000); + rt.real_epoch_offset_ns = 0; +} + +test "sleep(.none) is not a wait and not a cancelation point" { + const rt = testRuntime(); + const io = rt.io(); + rt.current.cancel_pending = true; + try io.vtable.sleep(io.userdata, .none); + try testing.expect(rt.current.cancel_pending); + rt.current.cancel_pending = false; +} + +test "random fills, is not all zero, and does not repeat itself" { + const rt = testRuntime(); + const io = rt.io(); + var a: [32]u8 = @splat(0); + var b: [32]u8 = @splat(0); + io.random(&a); + io.random(&b); + try testing.expect(!std.mem.allEqual(u8, &a, 0)); + try testing.expect(!std.mem.eql(u8, &a, &b)); + // The one thing this port will not pretend about. + try testing.expectError(error.EntropyUnavailable, io.randomSecure(&a)); +} + +test "a task's stack high-water mark is measurable" { + const rt = testRuntime(); + const io = rt.io(); + var f = io.async(add, .{ 1, 2 }); + _ = f.await(io); + // Slot 1 ran `add` through the trampoline; something was written, and nowhere near 8 KiB. + const used = rt.stackUsed(&rt.slots[0]); + try testing.expect(used > 0); + try testing.expect(used < 8 * 1024); + try testing.expectEqual(@as(usize, 0), rt.stackUsed(&rt.main)); +} + +fn childAcks(io: Io, trace: *Trace) void { + io.sleep(.fromMilliseconds(1000), .awake) catch { + trace.put('c'); + return; + }; + trace.put('C'); +} + +fn childNeverAcks(io: Io, trace: *Trace) void { + // Blocks, so the parent really has to wait for it, but never observes the cancelation - which + // is the case where the parent has to take its own request back. + const old = io.swapCancelProtection(.blocked); + io.sleep(.fromMilliseconds(5), .awake) catch unreachable; + _ = io.swapCancelProtection(old); + trace.put('C'); +} + +fn parentAwaits(io: Io, trace: *Trace, acks: bool) void { + var f = if (acks) io.async(childAcks, .{ io, trace }) else io.async(childNeverAcks, .{ io, trace }); + f.await(io); + trace.put('r'); + io.checkCancel() catch { + trace.put('x'); + return; + }; + trace.put('e'); +} + +test "a cancel that lands during await is forwarded to the child" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + var f = io.async(parentAwaits, .{ io, &trace, true }); + f.cancel(io); + + // 'c': the child's sleep reported `error.Canceled`, so the child consumed the request. + // 'r': the parent's `await` returned. 'e': and the parent's own next cancelation point is + // *clean*, because the cancelation was delivered into the child rather than to the parent. + try testing.expectEqualStrings("cre", trace.seen()); +} + +test "a cancel the child never acknowledges is re-armed on the parent" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + + var f = io.async(parentAwaits, .{ io, &trace, false }); + f.cancel(io); + + // 'C': the child ran to completion under cancel protection. 'r': await returned. 'x': and the + // request the parent gave away is back, so the parent's next cancelation point reports it - + // `Threaded.await` does the same thing through `recancelInner` (`Threaded.zig:2470`). + try testing.expectEqualStrings("Crx", trace.seen()); +} + +fn condWaiter(io: Io, m: *Io.Mutex, c: *Io.Condition, flag: *bool, trace: *Trace) void { + m.lock(io) catch return; + defer m.unlock(io); + while (!flag.*) c.wait(io, m) catch return; + trace.put('w'); +} + +fn condSignaler(io: Io, m: *Io.Mutex, c: *Io.Condition, flag: *bool, trace: *Trace) void { + m.lock(io) catch return; + flag.* = true; + m.unlock(io); + trace.put('s'); + c.signal(io); +} + +test "std.Io.Condition signals across tasks" { + const rt = testRuntime(); + const io = rt.io(); + var trace: Trace = .{}; + var m: Io.Mutex = .init; + var c: Io.Condition = .init; + var flag = false; + + // `Condition` is the most futex-dependent thing in std: an epoch word, a packed + // waiters/signals state, and a mutex handed back and forth across the wait. + var w = io.async(condWaiter, .{ io, &m, &c, &flag, &trace }); + var s = io.async(condSignaler, .{ io, &m, &c, &flag, &trace }); + w.await(io); + s.await(io); + + try testing.expectEqualStrings("sw", trace.seen()); + try testing.expect(m.tryLock()); +} diff --git a/src/io/unimplemented.zig b/src/io/unimplemented.zig new file mode 100644 index 0000000..566afac --- /dev/null +++ b/src/io/unimplemented.zig @@ -0,0 +1,490 @@ +//! Every `std.Io.VTable` entry this port does not implement, as a function that panics with its +//! own name. +//! +//! `std.Io.VTable` has 109 entries (`/usr/lib/zig/std/Io.zig:51-255`). A port of it to a chip with +//! no filesystem, no sockets and no processes implements sixteen of them, and the honest way to +//! express the other 93 is not a partial struct - the struct has no optional fields - but a complete +//! one whose unimplemented entries announce themselves the moment they are reached. So this file is +//! the floor: `p4.zig` starts from `unimplemented.vtable` and overwrites the entries it really has, +//! and anything it forgot fails with the name of the entry rather than a jump to address zero. +//! +//! A stub for an entry `p4.zig` does implement is dead the moment the assignment overwrites the +//! field, so it costs nothing: a riscv32 `-OReleaseSmall` build emits exactly the 93 that survive. +//! +//! The signatures are transcribed from that VTable declaration verbatim, which is why they carry +//! precision about types nothing here will ever construct: the compiler checks each one against +//! std, so a Zig release that changes a signature - or adds an entry - is a build error in this +//! file instead of a silent gap. The parameter names are dropped to `_` because a stub that reads +//! its arguments would be a lie. +//! +//! Note the five `fileMemoryMap*` entries. Their signatures name `Io.File.MemoryMap`, and merely +//! *defining* a function with that in its type forces the struct to be laid out, which on +//! riscv32-freestanding needs `std.options.page_size_min` - see `p4.zig`'s "Integrating this". + +const std = @import("std"); +const Io = std.Io; + +const Alignment = std.mem.Alignment; +const AnyFuture = Io.AnyFuture; +const Batch = Io.Batch; +const CancelProtection = Io.CancelProtection; +const Cancelable = Io.Cancelable; +const Clock = Io.Clock; +const ConcurrentError = Io.ConcurrentError; +const Dir = Io.Dir; +const Duration = Io.Duration; +const File = Io.File; +const Group = Io.Group; +const LockedStderr = Io.LockedStderr; +const Operation = Io.Operation; +const Queue = Io.Queue; +const RandomSecureError = Io.RandomSecureError; +const Terminal = Io.Terminal; +const Timeout = Io.Timeout; +const Timestamp = Io.Timestamp; +const net = Io.net; + +/// A `std.Io.VTable` in which nothing is implemented. Every entry panics with its own name. +pub const vtable: Io.VTable = .{ + .crashHandler = panics.crashHandler, + .async = panics.async, + .concurrent = panics.concurrent, + .await = panics.await, + .cancel = panics.cancel, + .groupAsync = panics.groupAsync, + .groupConcurrent = panics.groupConcurrent, + .groupAwait = panics.groupAwait, + .groupCancel = panics.groupCancel, + .recancel = panics.recancel, + .swapCancelProtection = panics.swapCancelProtection, + .checkCancel = panics.checkCancel, + .futexWait = panics.futexWait, + .futexWaitUncancelable = panics.futexWaitUncancelable, + .futexWake = panics.futexWake, + .operate = panics.operate, + .batchAwaitAsync = panics.batchAwaitAsync, + .batchAwaitConcurrent = panics.batchAwaitConcurrent, + .batchCancel = panics.batchCancel, + .dirCreateDir = panics.dirCreateDir, + .dirCreateDirPath = panics.dirCreateDirPath, + .dirCreateDirPathOpen = panics.dirCreateDirPathOpen, + .dirOpenDir = panics.dirOpenDir, + .dirStat = panics.dirStat, + .dirStatFile = panics.dirStatFile, + .dirAccess = panics.dirAccess, + .dirCreateFile = panics.dirCreateFile, + .dirCreateFileAtomic = panics.dirCreateFileAtomic, + .dirOpenFile = panics.dirOpenFile, + .dirClose = panics.dirClose, + .dirRead = panics.dirRead, + .dirRealPath = panics.dirRealPath, + .dirRealPathFile = panics.dirRealPathFile, + .dirDeleteFile = panics.dirDeleteFile, + .dirDeleteDir = panics.dirDeleteDir, + .dirRename = panics.dirRename, + .dirRenamePreserve = panics.dirRenamePreserve, + .dirSymLink = panics.dirSymLink, + .dirReadLink = panics.dirReadLink, + .dirSetOwner = panics.dirSetOwner, + .dirSetFileOwner = panics.dirSetFileOwner, + .dirSetPermissions = panics.dirSetPermissions, + .dirSetFilePermissions = panics.dirSetFilePermissions, + .dirSetTimestamps = panics.dirSetTimestamps, + .dirHardLink = panics.dirHardLink, + .fileStat = panics.fileStat, + .fileLength = panics.fileLength, + .fileClose = panics.fileClose, + .fileWritePositional = panics.fileWritePositional, + .fileWriteFileStreaming = panics.fileWriteFileStreaming, + .fileWriteFilePositional = panics.fileWriteFilePositional, + .fileReadPositional = panics.fileReadPositional, + .fileSeekBy = panics.fileSeekBy, + .fileSeekTo = panics.fileSeekTo, + .fileSync = panics.fileSync, + .fileIsTty = panics.fileIsTty, + .fileEnableAnsiEscapeCodes = panics.fileEnableAnsiEscapeCodes, + .fileSupportsAnsiEscapeCodes = panics.fileSupportsAnsiEscapeCodes, + .fileSetLength = panics.fileSetLength, + .fileSetOwner = panics.fileSetOwner, + .fileSetPermissions = panics.fileSetPermissions, + .fileSetTimestamps = panics.fileSetTimestamps, + .fileLock = panics.fileLock, + .fileTryLock = panics.fileTryLock, + .fileUnlock = panics.fileUnlock, + .fileDowngradeLock = panics.fileDowngradeLock, + .fileRealPath = panics.fileRealPath, + .fileHardLink = panics.fileHardLink, + .fileMemoryMapCreate = panics.fileMemoryMapCreate, + .fileMemoryMapDestroy = panics.fileMemoryMapDestroy, + .fileMemoryMapSetLength = panics.fileMemoryMapSetLength, + .fileMemoryMapRead = panics.fileMemoryMapRead, + .fileMemoryMapWrite = panics.fileMemoryMapWrite, + .processExecutableOpen = panics.processExecutableOpen, + .processExecutablePath = panics.processExecutablePath, + .lockStderr = panics.lockStderr, + .tryLockStderr = panics.tryLockStderr, + .unlockStderr = panics.unlockStderr, + .processCurrentPath = panics.processCurrentPath, + .processSetCurrentDir = panics.processSetCurrentDir, + .processSetCurrentPath = panics.processSetCurrentPath, + .processReplace = panics.processReplace, + .processReplacePath = panics.processReplacePath, + .processSpawn = panics.processSpawn, + .processSpawnPath = panics.processSpawnPath, + .childWait = panics.childWait, + .childKill = panics.childKill, + .progressParentFile = panics.progressParentFile, + .now = panics.now, + .clockResolution = panics.clockResolution, + .sleep = panics.sleep, + .random = panics.random, + .randomSecure = panics.randomSecure, + .netListenIp = panics.netListenIp, + .netAccept = panics.netAccept, + .netBindIp = panics.netBindIp, + .netConnectIp = panics.netConnectIp, + .netListenUnix = panics.netListenUnix, + .netConnectUnix = panics.netConnectUnix, + .netSocketCreatePair = panics.netSocketCreatePair, + .netSend = panics.netSend, + .netRead = panics.netRead, + .netWrite = panics.netWrite, + .netWriteFile = panics.netWriteFile, + .netClose = panics.netClose, + .netShutdown = panics.netShutdown, + .netInterfaceNameResolve = panics.netInterfaceNameResolve, + .netInterfaceName = panics.netInterfaceName, + .netLookup = panics.netLookup, +}; + +/// One function per entry. Grouped in a namespace so the panic message and the vtable field +/// cannot drift apart: `panics.dirOpenFile` is the entry named `dirOpenFile`. +const panics = struct { + fn crashHandler(_: ?*anyopaque) void { + @panic("io.p4 unimplemented: crashHandler"); + } + fn async(_: ?*anyopaque, _: []u8, _: std.mem.Alignment, _: []const u8, _: std.mem.Alignment, _: *const fn (context: *const anyopaque, result: *anyopaque) void) ?*AnyFuture { + @panic("io.p4 unimplemented: async"); + } + fn concurrent(_: ?*anyopaque, _: usize, _: std.mem.Alignment, _: []const u8, _: std.mem.Alignment, _: *const fn (context: *const anyopaque, result: *anyopaque) void) ConcurrentError!*AnyFuture { + @panic("io.p4 unimplemented: concurrent"); + } + fn await(_: ?*anyopaque, _: *AnyFuture, _: []u8, _: std.mem.Alignment) void { + @panic("io.p4 unimplemented: await"); + } + fn cancel(_: ?*anyopaque, _: *AnyFuture, _: []u8, _: std.mem.Alignment) void { + @panic("io.p4 unimplemented: cancel"); + } + fn groupAsync(_: ?*anyopaque, _: *Group, _: []const u8, _: std.mem.Alignment, _: *const fn (context: *const anyopaque) void) void { + @panic("io.p4 unimplemented: groupAsync"); + } + fn groupConcurrent(_: ?*anyopaque, _: *Group, _: []const u8, _: std.mem.Alignment, _: *const fn (context: *const anyopaque) void) ConcurrentError!void { + @panic("io.p4 unimplemented: groupConcurrent"); + } + fn groupAwait(_: ?*anyopaque, _: *Group, _: *anyopaque) Cancelable!void { + @panic("io.p4 unimplemented: groupAwait"); + } + fn groupCancel(_: ?*anyopaque, _: *Group, _: *anyopaque) void { + @panic("io.p4 unimplemented: groupCancel"); + } + fn recancel(_: ?*anyopaque) void { + @panic("io.p4 unimplemented: recancel"); + } + fn swapCancelProtection(_: ?*anyopaque, _: CancelProtection) CancelProtection { + @panic("io.p4 unimplemented: swapCancelProtection"); + } + fn checkCancel(_: ?*anyopaque) Cancelable!void { + @panic("io.p4 unimplemented: checkCancel"); + } + fn futexWait(_: ?*anyopaque, _: *const u32, _: u32, _: Timeout) Cancelable!void { + @panic("io.p4 unimplemented: futexWait"); + } + fn futexWaitUncancelable(_: ?*anyopaque, _: *const u32, _: u32) void { + @panic("io.p4 unimplemented: futexWaitUncancelable"); + } + fn futexWake(_: ?*anyopaque, _: *const u32, _: u32) void { + @panic("io.p4 unimplemented: futexWake"); + } + fn operate(_: ?*anyopaque, _: Operation) Cancelable!Operation.Result { + @panic("io.p4 unimplemented: operate"); + } + fn batchAwaitAsync(_: ?*anyopaque, _: *Batch) Cancelable!void { + @panic("io.p4 unimplemented: batchAwaitAsync"); + } + fn batchAwaitConcurrent(_: ?*anyopaque, _: *Batch, _: Timeout) Batch.AwaitConcurrentError!void { + @panic("io.p4 unimplemented: batchAwaitConcurrent"); + } + fn batchCancel(_: ?*anyopaque, _: *Batch) void { + @panic("io.p4 unimplemented: batchCancel"); + } + fn dirCreateDir(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.Permissions) Dir.CreateDirError!void { + @panic("io.p4 unimplemented: dirCreateDir"); + } + fn dirCreateDirPath(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.Permissions) Dir.CreateDirPathError!Dir.CreatePathStatus { + @panic("io.p4 unimplemented: dirCreateDirPath"); + } + fn dirCreateDirPathOpen(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.Permissions, _: Dir.OpenOptions) Dir.CreateDirPathOpenError!Dir { + @panic("io.p4 unimplemented: dirCreateDirPathOpen"); + } + fn dirOpenDir(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.OpenOptions) Dir.OpenError!Dir { + @panic("io.p4 unimplemented: dirOpenDir"); + } + fn dirStat(_: ?*anyopaque, _: Dir) Dir.StatError!Dir.Stat { + @panic("io.p4 unimplemented: dirStat"); + } + fn dirStatFile(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.StatFileOptions) Dir.StatFileError!File.Stat { + @panic("io.p4 unimplemented: dirStatFile"); + } + fn dirAccess(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.AccessOptions) Dir.AccessError!void { + @panic("io.p4 unimplemented: dirAccess"); + } + fn dirCreateFile(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.CreateFileOptions) File.OpenError!File { + @panic("io.p4 unimplemented: dirCreateFile"); + } + fn dirCreateFileAtomic(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.CreateFileAtomicOptions) Dir.CreateFileAtomicError!File.Atomic { + @panic("io.p4 unimplemented: dirCreateFileAtomic"); + } + fn dirOpenFile(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.OpenFileOptions) File.OpenError!File { + @panic("io.p4 unimplemented: dirOpenFile"); + } + fn dirClose(_: ?*anyopaque, _: []const Dir) void { + @panic("io.p4 unimplemented: dirClose"); + } + fn dirRead(_: ?*anyopaque, _: *Dir.Reader, _: []Dir.Entry) Dir.Reader.Error!usize { + @panic("io.p4 unimplemented: dirRead"); + } + fn dirRealPath(_: ?*anyopaque, _: Dir, _: []u8) Dir.RealPathError!usize { + @panic("io.p4 unimplemented: dirRealPath"); + } + fn dirRealPathFile(_: ?*anyopaque, _: Dir, _: []const u8, _: []u8) Dir.RealPathFileError!usize { + @panic("io.p4 unimplemented: dirRealPathFile"); + } + fn dirDeleteFile(_: ?*anyopaque, _: Dir, _: []const u8) Dir.DeleteFileError!void { + @panic("io.p4 unimplemented: dirDeleteFile"); + } + fn dirDeleteDir(_: ?*anyopaque, _: Dir, _: []const u8) Dir.DeleteDirError!void { + @panic("io.p4 unimplemented: dirDeleteDir"); + } + fn dirRename(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir, _: []const u8) Dir.RenameError!void { + @panic("io.p4 unimplemented: dirRename"); + } + fn dirRenamePreserve(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir, _: []const u8) Dir.RenamePreserveError!void { + @panic("io.p4 unimplemented: dirRenamePreserve"); + } + fn dirSymLink(_: ?*anyopaque, _: Dir, _: []const u8, _: []const u8, _: Dir.SymLinkFlags) Dir.SymLinkError!void { + @panic("io.p4 unimplemented: dirSymLink"); + } + fn dirReadLink(_: ?*anyopaque, _: Dir, _: []const u8, _: []u8) Dir.ReadLinkError!usize { + @panic("io.p4 unimplemented: dirReadLink"); + } + fn dirSetOwner(_: ?*anyopaque, _: Dir, _: ?File.Uid, _: ?File.Gid) Dir.SetOwnerError!void { + @panic("io.p4 unimplemented: dirSetOwner"); + } + fn dirSetFileOwner(_: ?*anyopaque, _: Dir, _: []const u8, _: ?File.Uid, _: ?File.Gid, _: Dir.SetFileOwnerOptions) Dir.SetFileOwnerError!void { + @panic("io.p4 unimplemented: dirSetFileOwner"); + } + fn dirSetPermissions(_: ?*anyopaque, _: Dir, _: Dir.Permissions) Dir.SetPermissionsError!void { + @panic("io.p4 unimplemented: dirSetPermissions"); + } + fn dirSetFilePermissions(_: ?*anyopaque, _: Dir, _: []const u8, _: File.Permissions, _: Dir.SetFilePermissionsOptions) Dir.SetFilePermissionsError!void { + @panic("io.p4 unimplemented: dirSetFilePermissions"); + } + fn dirSetTimestamps(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir.SetTimestampsOptions) Dir.SetTimestampsError!void { + @panic("io.p4 unimplemented: dirSetTimestamps"); + } + fn dirHardLink(_: ?*anyopaque, _: Dir, _: []const u8, _: Dir, _: []const u8, _: Dir.HardLinkOptions) Dir.HardLinkError!void { + @panic("io.p4 unimplemented: dirHardLink"); + } + fn fileStat(_: ?*anyopaque, _: File) File.StatError!File.Stat { + @panic("io.p4 unimplemented: fileStat"); + } + fn fileLength(_: ?*anyopaque, _: File) File.LengthError!u64 { + @panic("io.p4 unimplemented: fileLength"); + } + fn fileClose(_: ?*anyopaque, _: []const File) void { + @panic("io.p4 unimplemented: fileClose"); + } + fn fileWritePositional(_: ?*anyopaque, _: File, _: []const u8, _: []const []const u8, _: usize, _: u64) File.WritePositionalError!usize { + @panic("io.p4 unimplemented: fileWritePositional"); + } + fn fileWriteFileStreaming(_: ?*anyopaque, _: File, _: []const u8, _: *Io.File.Reader, _: Io.Limit) File.Writer.WriteFileError!usize { + @panic("io.p4 unimplemented: fileWriteFileStreaming"); + } + fn fileWriteFilePositional(_: ?*anyopaque, _: File, _: []const u8, _: *Io.File.Reader, _: Io.Limit, _: u64) File.WriteFilePositionalError!usize { + @panic("io.p4 unimplemented: fileWriteFilePositional"); + } + fn fileReadPositional(_: ?*anyopaque, _: File, _: []const []u8, _: u64) File.ReadPositionalError!usize { + @panic("io.p4 unimplemented: fileReadPositional"); + } + fn fileSeekBy(_: ?*anyopaque, _: File, _: i64) File.SeekError!void { + @panic("io.p4 unimplemented: fileSeekBy"); + } + fn fileSeekTo(_: ?*anyopaque, _: File, _: u64) File.SeekError!void { + @panic("io.p4 unimplemented: fileSeekTo"); + } + fn fileSync(_: ?*anyopaque, _: File) File.SyncError!void { + @panic("io.p4 unimplemented: fileSync"); + } + fn fileIsTty(_: ?*anyopaque, _: File) Cancelable!bool { + @panic("io.p4 unimplemented: fileIsTty"); + } + fn fileEnableAnsiEscapeCodes(_: ?*anyopaque, _: File) File.EnableAnsiEscapeCodesError!void { + @panic("io.p4 unimplemented: fileEnableAnsiEscapeCodes"); + } + fn fileSupportsAnsiEscapeCodes(_: ?*anyopaque, _: File) Cancelable!bool { + @panic("io.p4 unimplemented: fileSupportsAnsiEscapeCodes"); + } + fn fileSetLength(_: ?*anyopaque, _: File, _: u64) File.SetLengthError!void { + @panic("io.p4 unimplemented: fileSetLength"); + } + fn fileSetOwner(_: ?*anyopaque, _: File, _: ?File.Uid, _: ?File.Gid) File.SetOwnerError!void { + @panic("io.p4 unimplemented: fileSetOwner"); + } + fn fileSetPermissions(_: ?*anyopaque, _: File, _: File.Permissions) File.SetPermissionsError!void { + @panic("io.p4 unimplemented: fileSetPermissions"); + } + fn fileSetTimestamps(_: ?*anyopaque, _: File, _: File.SetTimestampsOptions) File.SetTimestampsError!void { + @panic("io.p4 unimplemented: fileSetTimestamps"); + } + fn fileLock(_: ?*anyopaque, _: File, _: File.Lock) File.LockError!void { + @panic("io.p4 unimplemented: fileLock"); + } + fn fileTryLock(_: ?*anyopaque, _: File, _: File.Lock) File.LockError!bool { + @panic("io.p4 unimplemented: fileTryLock"); + } + fn fileUnlock(_: ?*anyopaque, _: File) void { + @panic("io.p4 unimplemented: fileUnlock"); + } + fn fileDowngradeLock(_: ?*anyopaque, _: File) File.DowngradeLockError!void { + @panic("io.p4 unimplemented: fileDowngradeLock"); + } + fn fileRealPath(_: ?*anyopaque, _: File, _: []u8) File.RealPathError!usize { + @panic("io.p4 unimplemented: fileRealPath"); + } + fn fileHardLink(_: ?*anyopaque, _: File, _: Dir, _: []const u8, _: File.HardLinkOptions) File.HardLinkError!void { + @panic("io.p4 unimplemented: fileHardLink"); + } + fn fileMemoryMapCreate(_: ?*anyopaque, _: File, _: File.MemoryMap.CreateOptions) File.MemoryMap.CreateError!File.MemoryMap { + @panic("io.p4 unimplemented: fileMemoryMapCreate"); + } + fn fileMemoryMapDestroy(_: ?*anyopaque, _: *File.MemoryMap) void { + @panic("io.p4 unimplemented: fileMemoryMapDestroy"); + } + fn fileMemoryMapSetLength(_: ?*anyopaque, _: *File.MemoryMap, _: usize) File.MemoryMap.SetLengthError!void { + @panic("io.p4 unimplemented: fileMemoryMapSetLength"); + } + fn fileMemoryMapRead(_: ?*anyopaque, _: *File.MemoryMap) File.ReadPositionalError!void { + @panic("io.p4 unimplemented: fileMemoryMapRead"); + } + fn fileMemoryMapWrite(_: ?*anyopaque, _: *File.MemoryMap) File.WritePositionalError!void { + @panic("io.p4 unimplemented: fileMemoryMapWrite"); + } + fn processExecutableOpen(_: ?*anyopaque, _: Dir.OpenFileOptions) std.process.OpenExecutableError!File { + @panic("io.p4 unimplemented: processExecutableOpen"); + } + fn processExecutablePath(_: ?*anyopaque, _: []u8) std.process.ExecutablePathError!usize { + @panic("io.p4 unimplemented: processExecutablePath"); + } + fn lockStderr(_: ?*anyopaque, _: ?Terminal.Mode) Cancelable!LockedStderr { + @panic("io.p4 unimplemented: lockStderr"); + } + fn tryLockStderr(_: ?*anyopaque, _: ?Terminal.Mode) Cancelable!?LockedStderr { + @panic("io.p4 unimplemented: tryLockStderr"); + } + fn unlockStderr(_: ?*anyopaque) void { + @panic("io.p4 unimplemented: unlockStderr"); + } + fn processCurrentPath(_: ?*anyopaque, _: []u8) std.process.CurrentPathError!usize { + @panic("io.p4 unimplemented: processCurrentPath"); + } + fn processSetCurrentDir(_: ?*anyopaque, _: Dir) std.process.SetCurrentDirError!void { + @panic("io.p4 unimplemented: processSetCurrentDir"); + } + fn processSetCurrentPath(_: ?*anyopaque, _: []const u8) std.process.SetCurrentPathError!void { + @panic("io.p4 unimplemented: processSetCurrentPath"); + } + fn processReplace(_: ?*anyopaque, _: std.process.ReplaceOptions) std.process.ReplaceError { + @panic("io.p4 unimplemented: processReplace"); + } + fn processReplacePath(_: ?*anyopaque, _: Dir, _: std.process.ReplaceOptions) std.process.ReplaceError { + @panic("io.p4 unimplemented: processReplacePath"); + } + fn processSpawn(_: ?*anyopaque, _: std.process.SpawnOptions) std.process.SpawnError!std.process.Child { + @panic("io.p4 unimplemented: processSpawn"); + } + fn processSpawnPath(_: ?*anyopaque, _: Dir, _: std.process.SpawnOptions) std.process.SpawnError!std.process.Child { + @panic("io.p4 unimplemented: processSpawnPath"); + } + fn childWait(_: ?*anyopaque, _: *std.process.Child) std.process.Child.WaitError!std.process.Child.Term { + @panic("io.p4 unimplemented: childWait"); + } + fn childKill(_: ?*anyopaque, _: *std.process.Child) void { + @panic("io.p4 unimplemented: childKill"); + } + fn progressParentFile(_: ?*anyopaque) std.Progress.ParentFileError!File { + @panic("io.p4 unimplemented: progressParentFile"); + } + fn now(_: ?*anyopaque, _: Clock) Timestamp { + @panic("io.p4 unimplemented: now"); + } + fn clockResolution(_: ?*anyopaque, _: Clock) Clock.ResolutionError!Duration { + @panic("io.p4 unimplemented: clockResolution"); + } + fn sleep(_: ?*anyopaque, _: Timeout) Cancelable!void { + @panic("io.p4 unimplemented: sleep"); + } + fn random(_: ?*anyopaque, _: []u8) void { + @panic("io.p4 unimplemented: random"); + } + fn randomSecure(_: ?*anyopaque, _: []u8) RandomSecureError!void { + @panic("io.p4 unimplemented: randomSecure"); + } + fn netListenIp(_: ?*anyopaque, _: *const net.IpAddress, _: net.IpAddress.ListenOptions) net.IpAddress.ListenError!net.Socket { + @panic("io.p4 unimplemented: netListenIp"); + } + fn netAccept(_: ?*anyopaque, _: net.Socket.Handle, _: net.Server.AcceptOptions) net.Server.AcceptError!net.Socket { + @panic("io.p4 unimplemented: netAccept"); + } + fn netBindIp(_: ?*anyopaque, _: *const net.IpAddress, _: net.IpAddress.BindOptions) net.IpAddress.BindError!net.Socket { + @panic("io.p4 unimplemented: netBindIp"); + } + fn netConnectIp(_: ?*anyopaque, _: *const net.IpAddress, _: net.IpAddress.ConnectOptions) net.IpAddress.ConnectError!net.Socket { + @panic("io.p4 unimplemented: netConnectIp"); + } + fn netListenUnix(_: ?*anyopaque, _: *const net.UnixAddress, _: net.UnixAddress.ListenOptions) net.UnixAddress.ListenError!net.Socket.Handle { + @panic("io.p4 unimplemented: netListenUnix"); + } + fn netConnectUnix(_: ?*anyopaque, _: *const net.UnixAddress) net.UnixAddress.ConnectError!net.Socket.Handle { + @panic("io.p4 unimplemented: netConnectUnix"); + } + fn netSocketCreatePair(_: ?*anyopaque, _: net.Socket.CreatePairOptions) net.Socket.CreatePairError![2]net.Socket { + @panic("io.p4 unimplemented: netSocketCreatePair"); + } + fn netSend(_: ?*anyopaque, _: net.Socket.Handle, _: []net.OutgoingMessage, _: net.SendFlags) struct { ?net.Socket.SendError, usize } { + @panic("io.p4 unimplemented: netSend"); + } + fn netRead(_: ?*anyopaque, _: net.Socket.Handle, _: [][]u8) net.Stream.Reader.Error!usize { + @panic("io.p4 unimplemented: netRead"); + } + fn netWrite(_: ?*anyopaque, _: net.Socket.Handle, _: []const u8, _: []const []const u8, _: usize) net.Stream.Writer.Error!usize { + @panic("io.p4 unimplemented: netWrite"); + } + fn netWriteFile(_: ?*anyopaque, _: net.Socket.Handle, _: []const u8, _: *Io.File.Reader, _: Io.Limit) net.Stream.Writer.WriteFileError!usize { + @panic("io.p4 unimplemented: netWriteFile"); + } + fn netClose(_: ?*anyopaque, _: []const net.Socket.Handle) void { + @panic("io.p4 unimplemented: netClose"); + } + fn netShutdown(_: ?*anyopaque, _: net.Socket.Handle, _: net.ShutdownHow) net.ShutdownError!void { + @panic("io.p4 unimplemented: netShutdown"); + } + fn netInterfaceNameResolve(_: ?*anyopaque, _: *const net.Interface.Name) net.Interface.Name.ResolveError!net.Interface { + @panic("io.p4 unimplemented: netInterfaceNameResolve"); + } + fn netInterfaceName(_: ?*anyopaque, _: net.Interface) net.Interface.NameError!net.Interface.Name { + @panic("io.p4 unimplemented: netInterfaceName"); + } + fn netLookup(_: ?*anyopaque, _: net.HostName, _: *Queue(net.HostName.LookupResult), _: net.HostName.LookupOptions) net.HostName.LookupError!void { + @panic("io.p4 unimplemented: netLookup"); + } +}; diff --git a/src/main.zig b/src/main.zig new file mode 100644 index 0000000..2e65060 --- /dev/null +++ b/src/main.zig @@ -0,0 +1,104 @@ +//! Demo application: prove the toolchain, then blink. +//! +//! Nothing here is framework code. There is no FreeRTOS, no `app_main`, no libc and no +//! `compiler_rt`: `_start` is the reset entry the bootloader jumps to, the peripherals are the +//! comptime register model in soc.zig, and `printf` is a mask-ROM address. + +const std = @import("std"); +const config = @import("config"); +const soc = @import("soc"); + +const led: u6 = @intCast(config.led_pin); + +/// Zero-initialised statics live in L2MEM, which the bootloader does not clear. `_start` clears +/// them, and this array exists so the board can prove it happened. +export var bss_probe: [64]u32 = @splat(0); + +export fn zig_main() noreturn { + // readback: enable the pad's input buffer too, so the pin can be sampled while it is driven - + // which is what makes the blink self-verifying rather than hopeful. + soc.gpio.configureOutput(led, .{ .readback = true }); + + soc.rom.print("\r\nMARK ZIG_P4 toolchain=zig-only, no cmake, no esptool, no external linker\r\n", .{}); + soc.rom.print("MARK ZIG_P4_ROM ets_printf@0x%x reached from zig\r\n", .{@as(u32, @intFromPtr(&soc.rom.ets_printf))}); + + var bss_or: u32 = 0; + for (&bss_probe) |*w| bss_or |= @as(*volatile u32, w).*; + soc.rom.print("MARK ZIG_P4_BSS or-of-64-words=0x%08x expect=0x00000000\r\n", .{bss_or}); + + // A comptime hash of a compile-time string, checked against a value computed on the host: + // if the ABI or the linker script were wrong, this would not match. + const fnv = comptime fnv1a("0x4200.cafe"); + soc.rom.print("MARK ZIG_P4_FNV fnv1a(0x4200.cafe)=0x%08x expect=0x68440dea\r\n", .{fnv}); + + // A float the compiler cannot fold: loaded through a volatile pointer, so the arithmetic really + // happens at run time as fcvt.s.wu/fmul.s/fadd.s. Without the mstatus.FS write in _start this + // line is an unhandled illegal instruction - and with a constant the compiler would fold it + // away and the check would pass on a board whose FPU is still off. + var seven: u32 = 7; + var fx: f32 = @floatFromInt(@as(*volatile u32, &seven).*); + fx = fx * 1.5 + 0.25; + soc.rom.print("MARK ZIG_P4_FPU 7*1.5+0.25=%u.%u expect=10.75\r\n", .{ + @as(u32, @intFromFloat(fx)), @as(u32, @intFromFloat((fx - @trunc(fx)) * 100)), + }); + + const t0 = soc.cycles(); + soc.rom.ets_delay_us(1000); + const per_ms = soc.cycles() - t0; + soc.rom.print("MARK ZIG_P4_CLOCK %u cycles per ms\r\n", .{@as(u32, @intCast(per_ms))}); + + var beat: u32 = 0; + while (true) : (beat += 1) { + soc.gpio.setHigh(led); + const high = soc.gpio.getLevel(led); // sampled while driven high, so it proves the toggle + soc.rom.ets_delay_us(500_000); + soc.gpio.setLow(led); + const low = soc.gpio.getLevel(led); + soc.rom.ets_delay_us(500_000); + if (beat % 4 == 0) { + soc.rom.print("MARK ZIG_P4_ALIVE beat=%u gpio%u high=%u low=%u\r\n", .{ + beat, @as(u32, led), @as(u32, high), @as(u32, low), + }); + } + } +} + +fn fnv1a(comptime s: []const u8) u32 { + var h: u32 = 0x811c9dc5; + for (s) |b| { + h ^= b; + h = h *% 0x0100_0193; + } + return h; +} + +/// Reset entry. The bootloader hands over with an unspecified stack pointer and with the FPU +/// switched off, so: enable the F extension (mstatus.FS = Initial - the target includes `f`, and +/// ESP-IDF only ever enables the unit lazily from a trap handler, which this image does not have), +/// establish a stack in L2MEM, clear .bss, then call into Zig proper. +export fn _start() linksection(".text.entry") callconv(.naked) noreturn { + asm volatile ( + \\ li t0, 1 << 13 + \\ csrs mstatus, t0 + \\ la sp, __stack_top + \\ mv fp, sp + \\ la t0, __bss_start + \\ la t1, __bss_end + \\ bgeu t0, t1, 2f + \\1: + \\ sw zero, 0(t0) + \\ addi t0, t0, 4 + \\ bltu t0, t1, 1b + \\2: + \\ j zig_main + ); +} + +/// The current spelling: std wraps a bare three-argument `panic` in a compatibility shim whose own +/// comment calls it deprecated. +pub const panic = std.debug.FullPanic(struct { + fn call(msg: []const u8, _: ?usize) noreturn { + soc.rom.print("MARK ZIG_P4_PANIC %s\r\n", .{msg.ptr}); + while (true) {} + } +}.call); diff --git a/src/mmio.zig b/src/mmio.zig new file mode 100644 index 0000000..4e57985 --- /dev/null +++ b/src/mmio.zig @@ -0,0 +1,261 @@ +//! The typed layer over ESP-IDF's register macros. +//! +//! `@import("regs")` is `zig translate-c` over every `*_reg.h` header of the ESP32-P4: about 86,000 +//! flat constants, three per field - `X_REG` (address), `X_S` (shift), `X_V` (unshifted value mask). +//! Those are the right numbers but the wrong shape; writing `p.* = (p.* & ~(v << s)) | (x << s)` by +//! hand at every call site is how register bugs are made. +//! +//! This module turns those triples into checked accessors, at comptime, with no generated code: +//! +//! const conf0 = mmio.Reg.at(regs.LEDC_CH0_CONF0_REG); +//! const timer_sel = mmio.Field.of(regs.LEDC_TIMER_SEL_CH0_S, regs.LEDC_TIMER_SEL_CH0_V); +//! +//! conf0.set(timer_sel, 2); // read-modify-write just that field +//! conf0.modify(.{ timer_sel.is(2), en.is(1) }); // several fields, one store, rest preserved +//! conf0.write(.{ timer_sel.is(2), en.is(1) }); // several fields, one store, rest ZEROED +//! const t = conf0.get(timer_sel); +//! +//! **`modify` is the default; `write` is the exception.** The difference is what happens to the bits +//! you did not name, and on this chip that is usually not "nothing to worry about": 4,126 of the +//! 20,007 documented fields (20.6%) have a non-zero reset value, and within the headers a low-level +//! driver actually touches, 628 of 1,345 registers (46.7%) contain at least one. A `write` that +//! names two fields silently zeroes those, so it is correct only where the whole word is being +//! established deliberately. +//! +//! The converse hazard is narrower than it looks. Read-modify-write is only unsafe on +//! write-1-to-clear and read-to-clear bits - 139 of 5,365 registers, 28 of them in scope here - +//! because a plain self-clearing (`WT`) or write-only bit reads back as 0, so the read-modify-write +//! rewrites 0 and triggers nothing. The registers that genuinely need care are the interrupt-status +//! ones, and they are recognisable: `INT`, `ST`, `RAW` in the name. +//! +//! For the write-1-to-set/write-1-to-clear *alias* registers (`GPIO_OUT_W1TS_REG` and friends) +//! neither applies - the right operation is `writeRaw(mask)`, and the hardware does the rest. +//! +//! Everything here is `inline` and comptime-folded: a `set` of a constant field with a constant +//! value compiles to the same three instructions as the hand-written version, and a composed +//! `write` of constants compiles to a single `li`/`sw` pair. + +const std = @import("std"); + +/// A C macro value from translate-c (`c_int`, `c_uint`, comptime_int) as a u32 address, checked. +/// +/// translate-c types most of these as `c_int`, i.e. signed. Any register address that overflowed +/// into negative would silently become a wild pointer, so the cast is a comptime assertion instead. +pub inline fn addr(comptime macro: anytype) u32 { + comptime { + const v = @as(i64, macro); + if (v < 0 or v > 0xffff_ffff) @compileError(std.fmt.comptimePrint( + "register address {d} is not a 32-bit address - translate-c signedness or the wrong macro", + .{v}, + )); + return @intCast(v); + } +} + +/// One field of a register: where it sits and how wide it is. +/// +/// Built from the `_S` and `_V` macro pair. `_V` is the *unshifted* mask, so it must be +/// `2^width - 1`; anything else means the macro is not a field mask and the caller has picked up +/// the wrong constant (`_M`, the pre-shifted mask, is the usual mistake). +pub const Field = struct { + shift: u5, + width: u6, + + pub inline fn of(comptime shift_macro: anytype, comptime mask_macro: anytype) Field { + comptime { + const s = @as(i64, shift_macro); + const m = @as(i64, mask_macro); + if (s < 0 or s > 31) @compileError(std.fmt.comptimePrint("field shift {d} out of range", .{s})); + if (m <= 0) @compileError(std.fmt.comptimePrint("field mask {d} is not positive", .{m})); + const um: u64 = @intCast(m); + if (um & (um + 1) != 0) @compileError(std.fmt.comptimePrint( + "field mask 0x{x} is not 2^n-1 - this looks like a pre-shifted _M macro, not a _V mask", + .{um}, + )); + const width = 64 - @clz(um); + if (s + width > 32) @compileError(std.fmt.comptimePrint( + "field at bit {d} is {d} bits wide, which runs past bit 31", + .{ s, width }, + )); + return .{ .shift = @intCast(s), .width = @intCast(width) }; + } + } + + /// A single-bit field, for the `(BIT(n))` style macros that carry no separate `_S`/`_V` pair. + pub inline fn bit(comptime n: anytype) Field { + comptime { + const b = @as(i64, n); + if (b < 0 or b > 31) @compileError(std.fmt.comptimePrint("bit {d} out of range", .{b})); + return .{ .shift = @intCast(b), .width = 1 }; + } + } + + /// Mask in place, i.e. what `_M` would have been. + pub inline fn mask(self: Field) u32 { + return self.unshiftedMask() << self.shift; + } + + pub inline fn unshiftedMask(self: Field) u32 { + return if (self.width >= 32) 0xffff_ffff else (@as(u32, 1) << @intCast(self.width)) - 1; + } + + pub inline fn max(self: Field) u32 { + return self.unshiftedMask(); + } + + /// Pair this field with a value, for a composed `Reg.write`. + pub inline fn is(self: Field, value: u32) Value { + return .{ .field = self, .value = value }; + } +}; + +/// A field/value pair, the argument type of `Reg.write`. +pub const Value = struct { + field: Field, + value: u32, +}; + +/// A 32-bit MMIO register. +pub const Reg = struct { + address: usize, + + pub inline fn at(comptime macro: anytype) Reg { + return .{ .address = addr(macro) }; + } + + /// For registers the HAL reaches by computed address (per-channel strides). + pub inline fn atAddress(a: usize) Reg { + return .{ .address = a }; + } + + pub inline fn ptr(self: Reg) *volatile u32 { + return @ptrFromInt(self.address); + } + + pub inline fn raw(self: Reg) u32 { + return self.ptr().*; + } + + pub inline fn writeRaw(self: Reg, v: u32) void { + self.ptr().* = v; + } + + /// Read one field, shifted down. + pub inline fn get(self: Reg, f: Field) u32 { + return (self.raw() >> f.shift) & f.unshiftedMask(); + } + + /// Read-modify-write one field, preserving every other bit. The right default. Unsafe only on + /// write-1-to-clear / read-to-clear bits, i.e. interrupt-status registers. + pub inline fn set(self: Reg, f: Field, value: u32) void { + const p = self.ptr(); + p.* = (p.* & ~f.mask()) | ((value & f.unshiftedMask()) << f.shift); + } + + /// Compose one store from a tuple of `field.is(value)` pairs, **zeroing every bit not named**. + /// Use only when establishing a whole word deliberately; `modify` is what a driver usually + /// wants, because nearly half the registers here have a field whose reset value is not zero. + /// + /// Naming two fields that share a bit is asserted against: it means one of the two constants is + /// wrong, and the hardware would silently get whichever won. + pub inline fn write(self: Reg, values: anytype) void { + var acc: u32 = 0; + var seen: u32 = 0; + inline for (values) |v| { + const m = v.field.mask(); + // A debug assert rather than a compile error: the pairs carry runtime values, so the + // geometry is not always comptime-known at this point. It fires in host tests and in + // Debug builds, and costs nothing in ReleaseSmall. + std.debug.assert(seen & m == 0); + seen |= m; + acc |= (v.value & v.field.unshiftedMask()) << v.field.shift; + } + self.writeRaw(acc); + } + + /// Read-modify-write several fields in one store, leaving every other bit as it was. + pub inline fn modify(self: Reg, values: anytype) void { + var keep: u32 = 0xffff_ffff; + var acc: u32 = 0; + inline for (values) |v| { + const m = v.field.mask(); + std.debug.assert(keep & m != 0); + keep &= ~m; + acc |= (v.value & v.field.unshiftedMask()) << v.field.shift; + } + const p = self.ptr(); + p.* = (p.* & keep) | acc; + } + + /// Spin until a field reads the wanted value. Returns false on timeout rather than hanging: + /// a peripheral that never answers is a bug to report, not a board to power-cycle. + pub inline fn waitFor(self: Reg, f: Field, want: u32, spins: u32) bool { + var n: u32 = 0; + while (n < spins) : (n += 1) { + if (self.get(f) == want) return true; + } + return false; + } +}; + +/// An array of identical registers, for the per-channel blocks (LEDC channels, timer groups, UARTs) +/// whose macros come one-per-instance. The stride is checked against a second instance's macro, so +/// a wrong stride is a compile error rather than a wild write into the next channel. +pub fn RegArray(comptime first: anytype, comptime second: anytype, comptime count: u32) type { + return struct { + pub const base = addr(first); + pub const stride = addr(second) - addr(first); + pub const len = count; + + comptime { + if (addr(second) <= addr(first)) @compileError("RegArray: second instance is not above the first"); + } + + pub inline fn at(i: u32) Reg { + std.debug.assert(i < count); + return Reg.atAddress(base + stride * i); + } + }; +} + +test "field geometry is derived from the macro pair" { + const f = Field.of(5, 0x3ff); // LEDC_OVF_NUM_CH0: bitpos [14:5] + try std.testing.expectEqual(@as(u5, 5), f.shift); + try std.testing.expectEqual(@as(u6, 10), f.width); + try std.testing.expectEqual(@as(u32, 0x3ff << 5), f.mask()); + try std.testing.expectEqual(@as(u32, 1023), f.max()); +} + +test "single bit fields" { + const f = Field.bit(2); + try std.testing.expectEqual(@as(u32, 4), f.mask()); + try std.testing.expectEqual(@as(u6, 1), f.width); +} + +test "composed write builds one word" { + // The bit pattern a real LEDC channel enable would produce: timer_sel=2, sig_out_en=1. + const timer_sel = Field.of(0, 0x3); + const sig_out_en = Field.bit(2); + var cell: u32 = 0xffff_ffff; + const r = Reg.atAddress(@intFromPtr(&cell)); + r.write(.{ timer_sel.is(2), sig_out_en.is(1) }); + try std.testing.expectEqual(@as(u32, 0b110), cell); +} + +test "modify preserves unnamed bits, write does not" { + const lo = Field.of(0, 0xf); + var cell: u32 = 0xdead_beef; + const r = Reg.atAddress(@intFromPtr(&cell)); + r.modify(.{lo.is(0x5)}); + try std.testing.expectEqual(@as(u32, 0xdead_bee5), cell); + r.write(.{lo.is(0x5)}); + try std.testing.expectEqual(@as(u32, 0x5), cell); +} + +test "values wider than the field are truncated, not smeared into neighbours" { + const f = Field.of(4, 0xf); + var cell: u32 = 0; + const r = Reg.atAddress(@intFromPtr(&cell)); + r.set(f, 0xff); + try std.testing.expectEqual(@as(u32, 0xf0), cell); +} diff --git a/src/net/all.zig b/src/net/all.zig new file mode 100644 index 0000000..3fc6dba --- /dev/null +++ b/src/net/all.zig @@ -0,0 +1,156 @@ +//! Everything the radio path needs, in one translation unit. +//! +//! This file exists for the same reason src/appdesc.zig is a separate object: the files it names +//! define `export`ed C symbols that ESP-Hosted's C calls, and nothing in the application source +//! mentions them. An ordinary `@import` would be analysed lazily under ReleaseSmall, the exports +//! would never be emitted, and the link would fail with a list of missing `_h_*` symbols that looks +//! like the port table was never written. +//! +//! Referencing each import in a `comptime` block forces analysis, which forces the exports. +//! +//! The layers, bottom up: +//! +//! src/hal/sdmmc.zig the P4's SDMMC peripheral as an SDIO host +//! src/io/p4.zig std.Io for this chip - the cooperative runtime everything above uses +//! src/net/libc.zig the libc symbols ESP-Hosted's C reaches for +//! src/net/port.zig `g_h`, the table ESP-Hosted reaches the machine through +//! src/net/hosted_glue.zig logging, event bases, and loud stubs for layers not yet run +//! src/net/ip.zig IPv4/ARP/ICMP/UDP/DHCP/TCP/HTTP, replacing lwIP +//! src/net/link.zig ESP-Hosted's station channel, bridged to that stack +//! +//! ESP-Hosted's transport C sits on top of `port.zig` and is compiled by `hostedC` in build.zig. + +const std = @import("std"); + +pub const libc = @import("libc.zig"); +pub const glue = @import("hosted_glue.zig"); +pub const port = @import("port.zig"); +pub const heap = @import("heap.zig"); +pub const os = @import("hosted_os.zig"); +pub const ip = @import("ip.zig"); +/// The station data path. Separate from `init` on purpose: bringing the transport up and putting an +/// IP stack on the station interface are two decisions, and an application may want the first +/// without the second (examples/radio.zig does). Call `link.open()` once `hosted_wifi_sta_start` +/// has returned. +pub const link = @import("link.zig"); +/// The std.Io implementation. A module rather than a path: Zig confines a module's imports to its +/// own root directory, so src/net/ cannot reach ../io/ by file. build.zig wires it as `io`. +pub const runtime = @import("io"); + +comptime { + _ = libc; + _ = glue; + _ = port; + _ = heap; + _ = os; + _ = ip; + _ = link; + _ = runtime; +} + +/// ESP-Hosted's transport entry point, from +/// host/drivers/transport/transport_drv.h:131. Returns an `esp_err_t`, so 0 is success. The +/// callback fires once the slave has answered and the transport has reached its "active" state, +/// which is the moment the radio becomes usable. +extern fn setup_transport(up_cb: ?*const fn () callconv(.c) void) c_int; + +/// Populate the transport configuration from Kconfig - SDIO slot, bus width, clock, the pin map and +/// the C6 reset pin. From host/api/src/esp_hosted_transport_config.c:25. +/// +/// This is not optional and skipping it does not fail loudly. `esp_hosted_sdio_get_config` hands +/// back a pointer to static storage which starts out all zeroes, so a transport started without +/// this reads slot 0, width 0, 0 kHz, every pin GPIO0 and queue sizes of zero. The first run of +/// examples/radio.zig did exactly that: the only hint was ESP-Hosted warning "provided sdio tx queue +/// size is zero! Setting to 20", and then nothing ever came up. ESP-Hosted's own esp_hosted_init +/// calls this at esp_hosted_api.c:151; this file calls the same function for the same reason. +extern fn esp_hosted_set_default_config() c_int; + +/// True if a configuration has already been set, so `init` can be called twice without clobbering +/// a configuration an application deliberately overrode. +extern fn esp_hosted_is_config_valid() bool; + +/// Attempt the connection to the coprocessor: release its reset, bring the SDIO card up, and run +/// the capability handshake. From host/drivers/transport/transport_drv.h:133. +/// +/// `setup_transport` does NOT do this - it only calls transport_drv_init (bus and threads) and +/// stores the up-callback (transport_drv.c:170-177). ESP-Hosted's own API splits the two the same +/// way: esp_hosted_init calls setup_transport, and esp_hosted_connect_to_slave calls this +/// (esp_hosted_api.c:184). Calling only the first is a transport that exists and never speaks; that +/// is exactly what the third run of examples/radio.zig showed - threads created, nothing allocated, +/// no SDIO traffic, and silence for ten seconds. +extern fn transport_drv_reconfigure() c_int; + +/// Bring the RPC layer up and register its event callbacks. From +/// host/drivers/rpc/wrap/rpc_wrap.h:45,50. +/// +/// Separate from the transport on purpose: the transport is the pipe, RPC is the language spoken +/// over it. esp_hosted_init calls setup_transport and then these two (esp_hosted_api.c:154-156). +/// Skipping them leaves a transport that is genuinely up and a control path that answers every +/// request with "RPC not initialized or transport down, failing fast" - which is what the first +/// Wi-Fi call on this board printed. +/// +/// These must run BEFORE `transport_drv_reconfigure`, and the reason is a single line in +/// ESP-Hosted: `rpc_core_init` ends with `set_rpc_lib_state(RPC_LIB_STATE_INIT)` +/// (rpc_core.c:1164), and the *only* thing that ever raises that state to READY is `rpc_start`, +/// called from `transport_delayed_init` (transport_drv.c:802) on the transport's own RX thread the +/// moment the slave's INIT event is parsed. Call `rpc_init` after the transport is up and +/// `rpc_core_init` stamps INIT over the READY that already happened, with no second writer: both +/// `rpc_rx_thread` and `rpc_tx_thread` then sit in `if (!is_rpc_lib_ready()) _h_sleep(1)` +/// (rpc_core.c:482-485, :543-547) forever. `rpc_send_req` still succeeds - it only enqueues +/// (rpc_core.c:1019) - so every synchronous request is accepted, never transmitted, and returns +/// "Timeout waiting for Resp" ten seconds later. That is exactly the Req_WifiInit failure. +extern fn rpc_init() c_int; +extern fn rpc_register_event_callbacks() c_int; + +/// Set once the C reports the transport up. Read through `isUp`. +var transport_up: bool = false; + +fn onTransportUp() callconv(.c) void { + transport_up = true; +} + +/// True once the C6 has answered and ESP-Hosted's transport has reached its active state. +pub fn isUp() bool { + return transport_up; +} + +pub const Error = error{ ConfigFailed, TransportSetupFailed, SlaveConnectFailed, RpcInitFailed }; + +/// Bring the radio path up, in the one order that works. +/// +/// Each step depends on the one before it: +/// 1. the libc allocator must exist before ESP-Hosted allocates anything, and its very first act +/// is to allocate, +/// 2. the port table must be installed before the transport starts, because the transport reaches +/// the SDIO bus, the clock and its own threads through that table, +/// 3. the transport must be *set up* - bus, queues, threads - before RPC, because `rpc_core_init` +/// opens a serial endpoint on it, +/// 4. RPC must be initialised before the coprocessor is spoken to, because the handshake's own +/// `rpc_start` is what takes the RPC lib from INIT to READY and `rpc_core_init` would +/// overwrite it. See `rpc_init` above, +/// 5. only then is the C6 reset released and the capability handshake run, through our driver. +/// +/// Blocks for the handshake: step 5 is `transport_drv_reconfigure`, which polls the slave every +/// 200 ms (transport_drv.c:219-234). It runs on whatever task calls this, so that task must not be +/// the one printing progress. +pub fn init(io_impl: std.Io, gpa: std.mem.Allocator) Error!void { + libc.install(gpa); + port.install(io_impl, gpa); + // The configuration must exist before the transport reads it. Only set defaults if an + // application has not already provided its own, which is the order esp_hosted_init uses. + if (!esp_hosted_is_config_valid()) { + if (esp_hosted_set_default_config() != 0) return error.ConfigFailed; + } + if (setup_transport(&onTransportUp) != 0) return error.TransportSetupFailed; + + // The control path, before the pipe is opened. This is ESP-Hosted's own order + // (esp_hosted_api.c:154-156 init, then esp_hosted_api.c:184 connect) and it is load-bearing, + // not cosmetic: see the comment on `rpc_init` above. With RPC initialised first, `rpc_start` + // from the handshake is the last writer of the RPC lib state, and the reader and writer threads + // leave their not-ready loop for good. + if (rpc_init() != 0) return error.RpcInitFailed; + if (rpc_register_event_callbacks() != 0) return error.RpcInitFailed; + + // And now actually talk to the coprocessor. Blocks until the slave answers. + if (transport_drv_reconfigure() != 0) return error.SlaveConnectFailed; +} diff --git a/src/net/heap.zig b/src/net/heap.zig new file mode 100644 index 0000000..73ebcf2 --- /dev/null +++ b/src/net/heap.zig @@ -0,0 +1,673 @@ +//! Two allocators, because ESP-Hosted needs two different things and `std` supplies neither. +//! +//! **`Heap`** is a general-purpose allocator over a caller-supplied static buffer, exposed through +//! the ordinary `std.mem.Allocator` vtable. Writing one was not the first choice. `std.heap` was +//! read first, and nothing in it fits this workload: +//! +//! * `FixedBufferAllocator` can only free the *most recent* allocation. ESP-Hosted frees +//! per-packet buffers in whatever order the radio finishes with them. +//! * `ArenaAllocator` cannot reclaim at all until the whole arena dies, and this arena never dies. +//! * `BrkAllocator` (`std/heap/BrkAllocator.zig:38`) rounds its backing store to +//! `@max(64 * 1024, page_size_max)` per *size class*. One 64 KiB page per class does not fit in +//! a 128 KiB L2MEM, let alone the ~48 KiB this heap is meant to occupy. +//! * `DebugAllocator` is page-granular and carries per-page metadata for a debugging feature set +//! nothing here wants. +//! * `SmpAllocator`, `PageAllocator`, `c_allocator` all need an OS. +//! +//! So `Heap` is a K&R-style allocator: one address-sorted free list, coalescing on free, first fit. +//! The workload it is sized for is the one the parent measured - `mempool.c` recycling fixed-size +//! buffers - which is exactly the pattern that keeps a coalescing free list short and its first fit +//! O(1) in practice: freed blocks of one size are handed straight back out. +//! +//! **`CHeap`** is the part that has nothing to do with which allocator is underneath. C's `free` +//! takes a bare pointer and no length, and `std.mem.Allocator.rawFree` requires both the exact +//! length and the original alignment. `CHeap` recovers them from an eight-byte header stored +//! immediately below every pointer it hands out. That header is the price of the C ABI and it is +//! paid per allocation: +//! +//! _h_malloc(n) -> 8 bytes overhead +//! _h_malloc_align(n, 64) -> 64 bytes overhead (the header plus the alignment slack) +//! +//! `CHeap` is written against `std.mem.Allocator`, not against `Heap`, so `install()` can be handed +//! any allocator at all and the C side does not change. + +const std = @import("std"); +const assert = std.debug.assert; +const Allocator = std.mem.Allocator; +const Alignment = std.mem.Alignment; + +// ---------------------------------------------------------------------------------------- Heap + +/// A coalescing free-list allocator over one contiguous buffer. +/// +/// Not thread-safe, and deliberately so: this runs under a cooperative single-core scheduler where +/// no task can be preempted between two instructions, and every entry point here runs to completion +/// without yielding. An interrupt handler must never allocate - see `port.zig`, which never does. +pub const Heap = struct { + base: [*]align(granule) u8, + /// Arena length in bytes, always a multiple of `granule`. + len: u32, + /// Offset of the first free block's header, or `null_off`. + free_head: u32, + + /// Every block header and every payload is 8-byte aligned. Eight is `_Alignof(max_align_t)` on + /// rv32 (`long long` and `double` are 8-byte aligned), which is the weakest guarantee C's + /// `malloc` is allowed to make, so it is also the strongest one a caller may assume. + pub const granule = 8; + + /// Free blocks store their list link in the payload, so a block must hold a header plus one + /// link. Any split that would leave less than this is absorbed into the neighbour instead. + const min_block = @sizeOf(Block) + granule; + const null_off: u32 = std.math.maxInt(u32); + + /// Header of every block, allocated or free. + /// + /// `next` is only meaningful while the block is on the free list; in an allocated block it + /// holds `alloc_magic`, which turns a double free or a wild pointer into an assertion instead + /// of a corrupted list. + const Block = extern struct { + /// Total bytes of this block *including* the header. Always a multiple of `granule`. + size: u32, + next: u32, + + const alloc_magic: u32 = 0xA110_C8ED; + }; + + comptime { + assert(@sizeOf(Block) == granule); + assert(@alignOf(Block) <= granule); + } + + /// Take ownership of `buffer`. The whole buffer becomes one free block; nothing else is stored + /// outside it, so the heap's own footprint is `@sizeOf(Heap)` (12 bytes) plus the buffer. + pub fn init(buffer: []align(granule) u8) Heap { + const usable: u32 = @intCast(buffer.len & ~@as(usize, granule - 1)); + assert(usable >= min_block); + var h: Heap = .{ .base = buffer.ptr, .len = usable, .free_head = 0 }; + const first = h.blockAt(0); + first.* = .{ .size = usable, .next = null_off }; + return h; + } + + pub fn allocator(h: *Heap) Allocator { + return .{ .ptr = h, .vtable = &.{ + .alloc = alloc, + .resize = resize, + .remap = remap, + .free = freeFn, + } }; + } + + inline fn blockAt(h: *Heap, off: u32) *Block { + assert(off + @sizeOf(Block) <= h.len); + return @ptrCast(@alignCast(h.base + off)); + } + + inline fn payloadOf(h: *Heap, off: u32) [*]u8 { + return h.base + off + @sizeOf(Block); + } + + fn alloc(ctx: *anyopaque, len: usize, alignment: Alignment, ret_addr: usize) ?[*]u8 { + _ = ret_addr; + const h: *Heap = @ptrCast(@alignCast(ctx)); + + // A zero-length allocation still needs a distinct address with a valid header, because it + // will be handed back to `free` with its length and must be findable. + const want: u32 = std.math.cast(u32, std.mem.alignForward(usize, @max(len, granule), granule)) orelse return null; + const a: u32 = @intCast(@max(granule, alignment.toByteUnits())); + + var prev: u32 = null_off; + var cur: u32 = h.free_head; + while (cur != null_off) { + const blk = h.blockAt(cur); + + // Where the payload would land if the block were used as-is, and how far it has to + // move to satisfy `a`. A gap smaller than `min_block` cannot become its own free + // block, so step one whole alignment further - the block was sized for that case. + const natural = @intFromPtr(h.payloadOf(cur)); + var gap: u32 = @intCast(std.mem.alignForward(usize, natural, a) - natural); + if (gap != 0 and gap < min_block) gap += a; + + const need = gap + @sizeOf(Block) + want; + if (blk.size < need) { + prev = cur; + cur = blk.next; + continue; + } + + // The block that will be handed out starts `gap` bytes into the free block. When + // `gap` is zero that is the free block itself, which then leaves the list. + const alloc_off = cur + gap; + var alloc_size = blk.size - gap; + const tail_off = alloc_off + @sizeOf(Block) + want; + const tail_size = alloc_size - @sizeOf(Block) - want; + + if (gap == 0) { + h.unlink(prev, cur); + } else { + // The leading gap stays a free block at the same address, so the list order is + // unchanged and no relinking is needed. + blk.size = gap; + } + + if (tail_size >= min_block) { + alloc_size -= tail_size; + const tail = h.blockAt(tail_off); + tail.* = .{ .size = tail_size, .next = undefined }; + h.insert(tail_off); + } + + const out = h.blockAt(alloc_off); + out.* = .{ .size = alloc_size, .next = Block.alloc_magic }; + const payload = h.payloadOf(alloc_off); + assert(@intFromPtr(payload) % a == 0); + return payload; + } + return null; + } + + fn resize(ctx: *anyopaque, memory: []u8, alignment: Alignment, new_len: usize, ret_addr: usize) bool { + _ = alignment; + _ = ret_addr; + const h: *Heap = @ptrCast(@alignCast(ctx)); + const off = h.offsetOfPayload(memory.ptr); + const blk = h.blockAt(off); + assert(blk.next == Block.alloc_magic); + + const want: u32 = std.math.cast(u32, std.mem.alignForward(usize, @max(new_len, granule), granule)) orelse return false; + const have = blk.size - @sizeOf(Block); + if (want <= have) { + // Shrink: give the tail back if it is big enough to be a block of its own. + const tail_size = have - want; + if (tail_size >= min_block) { + blk.size -= tail_size; + const tail_off = off + @sizeOf(Block) + want; + h.blockAt(tail_off).* = .{ .size = tail_size, .next = undefined }; + h.insert(tail_off); + } + return true; + } + + // Grow in place only by swallowing the physically adjacent free block, which is the case + // that matters: `serial_ll_if.c:282` reassembles a fragmented RPC response by repeatedly + // reallocating the same buffer upward with nothing allocated after it. + const next_off = off + blk.size; + if (next_off >= h.len) return false; + const prev_link = h.findFreePredecessor(next_off) orelse return false; + const next = h.blockAt(next_off); + if (blk.size + next.size < @sizeOf(Block) + want) return false; + + h.unlink(prev_link, next_off); + blk.size += next.size; + const tail_size = blk.size - @sizeOf(Block) - want; + if (tail_size >= min_block) { + blk.size -= tail_size; + const tail_off = off + @sizeOf(Block) + want; + h.blockAt(tail_off).* = .{ .size = tail_size, .next = undefined }; + h.insert(tail_off); + } + return true; + } + + fn remap(ctx: *anyopaque, memory: []u8, alignment: Alignment, new_len: usize, ret_addr: usize) ?[*]u8 { + // Relocation is never cheaper here than the caller's own alloc/copy/free, because this + // allocator cannot move a block without copying it either. + return if (resize(ctx, memory, alignment, new_len, ret_addr)) memory.ptr else null; + } + + fn freeFn(ctx: *anyopaque, memory: []u8, alignment: Alignment, ret_addr: usize) void { + _ = alignment; + _ = ret_addr; + const h: *Heap = @ptrCast(@alignCast(ctx)); + const off = h.offsetOfPayload(memory.ptr); + // A double free lands here with `next` already holding a list offset rather than the + // magic, and would otherwise splice the block into the free list twice. + assert(h.blockAt(off).next == Block.alloc_magic); + h.insert(off); + } + + fn offsetOfPayload(h: *Heap, p: [*]u8) u32 { + const delta = @intFromPtr(p) - @intFromPtr(h.base); + assert(delta >= @sizeOf(Block) and delta < h.len); + return @intCast(delta - @sizeOf(Block)); + } + + /// Splice `off` out of the free list. `prev` is its predecessor, or `null_off` if it is head. + fn unlink(h: *Heap, prev: u32, off: u32) void { + const nxt = h.blockAt(off).next; + if (prev == null_off) h.free_head = nxt else h.blockAt(prev).next = nxt; + } + + /// The free-list predecessor of `off`, or null if `off` is not on the free list at all. + /// `null_off` is returned when `off` is the head, mirroring `unlink`'s convention. + fn findFreePredecessor(h: *Heap, off: u32) ?u32 { + var prev: u32 = null_off; + var cur = h.free_head; + while (cur != null_off) : ({ + prev = cur; + cur = h.blockAt(cur).next; + }) { + if (cur == off) return prev; + if (cur > off) return null; + } + return null; + } + + /// Insert a block into the address-sorted free list, coalescing with either neighbour it + /// physically touches. Address order is what makes coalescing a pointer comparison rather than + /// a search, and it is why the list is sorted at all. + fn insert(h: *Heap, off: u32) void { + var prev: u32 = null_off; + var cur = h.free_head; + while (cur != null_off and cur < off) : ({ + prev = cur; + cur = h.blockAt(cur).next; + }) {} + + const blk = h.blockAt(off); + blk.next = cur; + if (prev == null_off) h.free_head = off else h.blockAt(prev).next = off; + + if (cur != null_off and off + blk.size == cur) { + const nxt = h.blockAt(cur); + blk.size += nxt.size; + blk.next = nxt.next; + } + if (prev != null_off) { + const p = h.blockAt(prev); + if (prev + p.size == off) { + p.size += blk.size; + p.next = blk.next; + } + } + } + + pub const Stats = struct { + /// Bytes in the arena, header overhead included. + total: u32, + /// Bytes on the free list, header overhead included. + free: u32, + /// Largest single free block, which is the largest allocation that can still succeed + /// (less one header, less alignment slack). + largest_free: u32, + free_blocks: u32, + }; + + pub fn stats(h: *Heap) Stats { + var s: Stats = .{ .total = h.len, .free = 0, .largest_free = 0, .free_blocks = 0 }; + var cur = h.free_head; + while (cur != null_off) : (cur = h.blockAt(cur).next) { + const size = h.blockAt(cur).size; + s.free += size; + s.free_blocks += 1; + if (size > s.largest_free) s.largest_free = size; + } + return s; + } + + /// Walk the free list and assert every invariant. Used by the tests; also usable from a + /// hardware self-test, where a corrupted list is otherwise invisible until it is fatal. + pub fn check(h: *Heap) void { + var cur = h.free_head; + var prev: u32 = null_off; + while (cur != null_off) { + const blk = h.blockAt(cur); + assert(blk.size >= min_block); + assert(blk.size % granule == 0); + assert(cur % granule == 0); + assert(cur + blk.size <= h.len); + if (prev != null_off) { + // Sorted, and never two free blocks that touch: `insert` would have merged them. + assert(prev < cur); + assert(prev + h.blockAt(prev).size < cur); + } + prev = cur; + cur = blk.next; + } + } +}; + +// --------------------------------------------------------------------------------------- CHeap + +/// C `malloc`/`free`/`realloc` semantics on top of any `std.mem.Allocator`. +/// +/// The whole reason this type exists is that `free(p)` carries no size and `rawFree` demands one. +/// Every pointer handed to C therefore has a `Header` in the eight bytes below it, holding what +/// `rawFree` needs: the exact length that was allocated, and the distance back to the base pointer. +/// +/// The alignment passed to the backing allocator is always `granule` (8). Stronger alignments are +/// satisfied *inside* the allocation by over-allocating and moving the payload up, rather than by +/// asking the backing allocator for them - which keeps the header immediately below the payload in +/// every case, and means `Heap` only ever sees one alignment. +pub const CHeap = struct { + gpa: Allocator, + + /// Live bytes as seen by C, i.e. what was asked for, not what was consumed. `bytes_reserved` + /// is the honest number. + bytes_live: usize = 0, + bytes_reserved: usize = 0, + peak_reserved: usize = 0, + blocks_live: usize = 0, + /// Allocations that returned NULL. Nonzero means the heap is too small; ESP-Hosted logs and + /// limps on rather than failing loudly, so this counter is the only durable evidence. + failures: usize = 0, + + pub const granule = Heap.granule; + + const Header = extern struct { + /// Bytes passed to `rawAlloc`, and therefore the length `rawFree` must be given. + total: u32, + /// `payload - base`. Between `granule` and the requested alignment, inclusive. + offset: u16, + /// `log2` of the alignment C asked for. Kept for `realloc`, which must preserve it. + log2_align: u8, + magic: u8, + + const value: u8 = 0x48; // 'H' + }; + + comptime { + assert(@sizeOf(Header) == granule); + assert(@alignOf(Header) <= granule); + } + + /// The strongest alignment expressible in `Header.offset`. ESP-Hosted asks for at most 64 + /// (`HOSTED_MEM_ALIGNMENT_64`, port_esp_hosted_host_os.h:95). + pub const max_alignment = 1 << 15; + + pub fn malloc(c: *CHeap, size: usize) ?[*]u8 { + return c.mallocAligned(size, granule); + } + + /// `size` is rounded up to a multiple of `alignment` before allocating, which is what + /// `heap_caps_aligned_alloc` does and therefore what `_h_malloc_align`'s callers get today. + /// It matters for DMA: a buffer whose *end* is not aligned shares its last cache line with + /// whatever follows it. + pub fn mallocAligned(c: *CHeap, size: usize, alignment: usize) ?[*]u8 { + assert(std.math.isPowerOfTwo(alignment)); + assert(alignment <= max_alignment); + const a = @max(granule, alignment); + + const payload = std.mem.alignForward(usize, @max(size, 1), a); + // `a` bytes of slack is always enough: the base is `granule`-aligned, the header needs + // `granule` of that slack, and moving up to the next `a` boundary costs at most `a - + // granule` more. + const total = std.math.add(usize, payload, a) catch { + c.failures += 1; + return null; + }; + + const base = c.gpa.rawAlloc(total, .fromByteUnits(granule), @returnAddress()) orelse { + c.failures += 1; + return null; + }; + const user_addr = std.mem.alignForward(usize, @intFromPtr(base) + @sizeOf(Header), a); + const offset = user_addr - @intFromPtr(base); + assert(offset >= @sizeOf(Header) and offset <= a); + assert(offset + payload <= total); + + const user: [*]u8 = @ptrFromInt(user_addr); + headerOf(user).* = .{ + .total = @intCast(total), + .offset = @intCast(offset), + .log2_align = @intCast(std.math.log2_int(usize, a)), + .magic = Header.value, + }; + + c.bytes_live += size; + c.bytes_reserved += total; + c.blocks_live += 1; + if (c.bytes_reserved > c.peak_reserved) c.peak_reserved = c.bytes_reserved; + return user; + } + + pub fn calloc(c: *CHeap, count: usize, size: usize) ?[*]u8 { + const n = std.math.mul(usize, count, size) catch { + c.failures += 1; + return null; + }; + const p = c.malloc(n) orelse return null; + @memset(p[0..n], 0); + return p; + } + + pub fn free(c: *CHeap, ptr: ?[*]u8) void { + const user = ptr orelse return; + const h = headerOf(user).*; + assert(h.magic == Header.value); + const base: [*]u8 = @ptrFromInt(@intFromPtr(user) - h.offset); + // Poison the magic so a second free asserts here rather than corrupting the backing + // allocator's own bookkeeping several calls later. + headerOf(user).magic = 0; + + c.bytes_live -|= usableLen(h); + c.bytes_reserved -= h.total; + c.blocks_live -= 1; + c.gpa.rawFree(base[0..h.total], .fromByteUnits(granule), @returnAddress()); + } + + /// C `realloc`: null pointer means allocate, zero size means free, and the old contents are + /// preserved up to the smaller of the two sizes. + /// + /// Growth in place is attempted first. `serial_ll_if.c:282` reassembles a fragmented RPC + /// response by calling this in a loop on the same buffer, so a `realloc` that always copies + /// turns an n-fragment response into O(n^2) bytes moved. + pub fn realloc(c: *CHeap, ptr: ?[*]u8, new_size: usize) ?[*]u8 { + const user = ptr orelse return c.malloc(new_size); + if (new_size == 0) { + c.free(user); + return null; + } + + const h = headerOf(user).*; + assert(h.magic == Header.value); + const a = @as(usize, 1) << @intCast(h.log2_align); + const old_usable = usableLen(h); + if (new_size <= old_usable) return user; + + const base: [*]u8 = @ptrFromInt(@intFromPtr(user) - h.offset); + const new_total = std.math.add(usize, std.mem.alignForward(usize, new_size, a), a) catch { + c.failures += 1; + return null; + }; + if (c.gpa.rawResize(base[0..h.total], .fromByteUnits(granule), new_total, @returnAddress())) { + c.bytes_live += new_size - old_usable; + c.bytes_reserved += new_total - h.total; + if (c.bytes_reserved > c.peak_reserved) c.peak_reserved = c.bytes_reserved; + headerOf(user).total = @intCast(new_total); + return user; + } + + const fresh = c.mallocAligned(new_size, a) orelse return null; + @memcpy(fresh[0..old_usable], user[0..old_usable]); + c.free(user); + return fresh; + } + + /// Bytes the caller may legitimately touch. Larger than what was asked for whenever the + /// request was rounded up to the alignment. + fn usableLen(h: Header) usize { + return h.total - h.offset; + } + + inline fn headerOf(user: [*]u8) *Header { + return @ptrFromInt(@intFromPtr(user) - @sizeOf(Header)); + } +}; + +// ---------------------------------------------------------------------------------------- tests + +const testing = std.testing; + +fn testHeap(comptime bytes: usize) struct { buf: []align(Heap.granule) u8, heap: Heap } { + const buf = testing.allocator.alignedAlloc(u8, .fromByteUnits(Heap.granule), bytes) catch unreachable; + return .{ .buf = buf, .heap = Heap.init(buf) }; +} + +test "Heap: alloc, free, and reuse of a hole in the middle" { + var t = testHeap(4096); + defer testing.allocator.free(t.buf); + const a = t.heap.allocator(); + + const p0 = try a.alloc(u8, 64); + const p1 = try a.alloc(u8, 64); + const p2 = try a.alloc(u8, 64); + t.heap.check(); + + // Free the middle one. An arena or a FixedBufferAllocator cannot give this back; the whole + // point of this allocator is that the next 64-byte request lands right here. + a.free(p1); + t.heap.check(); + const p3 = try a.alloc(u8, 64); + try testing.expectEqual(p1.ptr, p3.ptr); + + a.free(p0); + a.free(p2); + a.free(p3); + t.heap.check(); + // Everything coalesced back into one block. + const s = t.heap.stats(); + try testing.expectEqual(@as(u32, 1), s.free_blocks); + try testing.expectEqual(s.total, s.free); +} + +test "Heap: the malloc/free/realloc churn that defeats an arena" { + var t = testHeap(16 * 1024); + defer testing.allocator.free(t.buf); + const a = t.heap.allocator(); + + // mempool.c's pattern: allocate a batch of same-size buffers, release them in a scrambled + // order, allocate the same batch again. An arena's high-water mark would double each round; + // this must not grow at all. + var live: [16][]u8 = undefined; + const order = [_]usize{ 7, 0, 15, 3, 11, 1, 9, 4, 13, 2, 8, 6, 14, 5, 12, 10 }; + + for (&live) |*slot| slot.* = try a.alloc(u8, 200); + const after_first_round = t.heap.stats().free; + + for (0..8) |_| { + for (order) |i| a.free(live[i]); + t.heap.check(); + for (&live) |*slot| slot.* = try a.alloc(u8, 200); + t.heap.check(); + try testing.expectEqual(after_first_round, t.heap.stats().free); + } + for (live) |slot| a.free(slot); + + // Interleave reallocs that grow past their block, which is the serial reassembly path. + var grow = try a.alloc(u8, 32); + @memset(grow, 0xAB); + var n: usize = 64; + while (n <= 2048) : (n *= 2) { + const old_len = grow.len; + grow = try a.realloc(grow, n); + try testing.expect(std.mem.allEqual(u8, grow[0..old_len], 0xAB)); + @memset(grow[old_len..], 0xAB); + t.heap.check(); + } + a.free(grow); + t.heap.check(); + try testing.expectEqual(t.heap.stats().total, t.heap.stats().free); +} + +test "Heap: strong alignment splits the leading gap back into the free list" { + var t = testHeap(8192); + defer testing.allocator.free(t.buf); + const a = t.heap.allocator(); + + // 64-byte alignment is what _h_malloc_align asks for on the SDIO data path. + var blocks: [8][]align(64) u8 = undefined; + for (&blocks, 0..) |*b, i| { + b.* = try a.alignedAlloc(u8, .@"64", 100 + i); + try testing.expectEqual(@as(usize, 0), @intFromPtr(b.ptr) % 64); + } + t.heap.check(); + for (blocks) |b| a.free(b); + t.heap.check(); + try testing.expectEqual(t.heap.stats().total, t.heap.stats().free); +} + +test "Heap: exhaustion returns null rather than trampling the arena" { + var t = testHeap(1024); + defer testing.allocator.free(t.buf); + const a = t.heap.allocator(); + + var held: [64][]u8 = undefined; + var n: usize = 0; + while (n < held.len) : (n += 1) { + held[n] = a.alloc(u8, 64) catch break; + } + try testing.expect(n > 0 and n < held.len); + try testing.expectError(error.OutOfMemory, a.alloc(u8, 64)); + t.heap.check(); + for (held[0..n]) |b| a.free(b); + t.heap.check(); + try testing.expectEqual(t.heap.stats().total, t.heap.stats().free); +} + +test "CHeap: malloc/free/realloc against the C ABI, over the Heap" { + var t = testHeap(16 * 1024); + defer testing.allocator.free(t.buf); + var c: CHeap = .{ .gpa = t.heap.allocator() }; + + const p = c.malloc(100).?; + @memset(p[0..100], 0x5A); + try testing.expectEqual(@as(usize, 0), @intFromPtr(p) % CHeap.granule); + try testing.expectEqual(@as(usize, 1), c.blocks_live); + + // realloc must preserve contents across a move. + const q = c.realloc(p, 4000).?; + try testing.expect(std.mem.allEqual(u8, q[0..100], 0x5A)); + // Shrinking inside the same block returns the same pointer, as C permits. + try testing.expectEqual(q, c.realloc(q, 8).?); + c.free(q); + try testing.expectEqual(@as(usize, 0), c.blocks_live); + try testing.expectEqual(@as(usize, 0), c.bytes_reserved); + + // calloc zeroes. + const z = c.calloc(10, 16).?; + try testing.expect(std.mem.allEqual(u8, z[0..160], 0)); + c.free(z); + + // free(NULL) is a no-op, and realloc(NULL, n) is malloc. + c.free(null); + const r = c.realloc(null, 32).?; + // realloc(p, 0) frees and yields NULL. + try testing.expectEqual(@as(?[*]u8, null), c.realloc(r, 0)); + try testing.expectEqual(@as(usize, 0), c.blocks_live); + + t.heap.check(); + try testing.expectEqual(t.heap.stats().total, t.heap.stats().free); +} + +test "CHeap: _h_malloc_align(n, 64) is 64-aligned at both ends and frees exactly" { + // 12 x (1536 rounded to 64, plus 64 of header and slack) = 19,200 bytes, plus block headers. + var t = testHeap(24 * 1024); + defer testing.allocator.free(t.buf); + var c: CHeap = .{ .gpa = t.heap.allocator() }; + + var held: [12][*]u8 = undefined; + for (&held, 0..) |*slot, i| { + slot.* = c.mallocAligned(1536 - i, 64).?; + try testing.expectEqual(@as(usize, 0), @intFromPtr(slot.*) % 64); + } + // 64-byte alignment costs exactly 64 bytes of overhead per buffer: the eight-byte header plus + // the slack that moves the payload onto the boundary. + try testing.expectEqual(@as(usize, 12), c.blocks_live); + for (held) |slot| c.free(slot); + try testing.expectEqual(@as(usize, 0), c.bytes_reserved); + t.heap.check(); + try testing.expectEqual(t.heap.stats().total, t.heap.stats().free); +} + +test "CHeap: allocation failure is reported, not fatal" { + var t = testHeap(1024); + defer testing.allocator.free(t.buf); + var c: CHeap = .{ .gpa = t.heap.allocator() }; + + try testing.expectEqual(@as(?[*]u8, null), c.malloc(100_000)); + try testing.expectEqual(@as(usize, 1), c.failures); + // The heap is untouched by the failure. + t.heap.check(); + try testing.expectEqual(t.heap.stats().total, t.heap.stats().free); +} diff --git a/src/net/hosted/abi_assert.c b/src/net/hosted/abi_assert.c new file mode 100644 index 0000000..8815e89 --- /dev/null +++ b/src/net/hosted/abi_assert.c @@ -0,0 +1,78 @@ +/* + * The one ABI check that stands between this build and a silent hang. + * + * `hosted_osi_funcs_t` (host/esp_hosted_os_abstraction.h) is the function-pointer table ESP-Hosted + * reaches everything through - memory, sync, threads, GPIO, SDIO. src/net/port.zig defines it in + * Zig. Zig can assert its own layout; it cannot assert C's. This file asserts C's, so the two are + * checked against each other at build time. + * + * Why this is not paranoia. Four of the table's entries are guarded: + * + * #ifdef H_USE_MEMPOOL <- host/esp_hosted_os_abstraction.h:64-69 + * void *(*_h_get_mempool)(...); + * ... + * #endif + * + * `#ifdef`, not `#if`. H_USE_MEMPOOL is defined by + * host/port/esp/freertos/include/port_esp_hosted_host_config.h:127-131 - to 1 or to 0, but always + * DEFINED. So the four pointers are present in any translation unit that saw that header, and + * absent in any that did not, and every entry after them shifts by four pointers. + * + * That is reachable, not theoretical: host/esp_hosted.h:14 and + * host/drivers/transport/transport_util.h:10 both include esp_hosted_os_abstraction.h as their + * FIRST include, so a TU reaching the struct through either of those - before any port header - + * gets the short layout. Under IDF's CMake the ordering happens to work out. Under our flags it + * would be luck. + * + * Measured with our exact flags, both ways: + * + * sizeof _h_config_gpio _h_event_post + * without the force-include 268 132 264 + * with the force-include 284 148 280 + * + * Sixteen bytes. A TU with the short layout calling _h_config_gpio jumps through a mempool + * pointer instead - which is a jump to the wrong function, on a board with no debugger, and the + * symptom would look exactly like the SDIO bus failing to come up. + * + * build.zig therefore force-includes port_esp_hosted_host_config.h into every ESP-Hosted + * translation unit, and compiles this file to assert that it worked. The numbers below are the + * long (correct) layout. + */ + +#include "esp_hosted_os_abstraction.h" +#include <stddef.h> + +/* The guard must be visible here, or this file is asserting the wrong layout and proving nothing. */ +#ifndef H_USE_MEMPOOL +#error "H_USE_MEMPOOL is not visible: the force-include of port_esp_hosted_host_config.h is missing." +#endif + +_Static_assert( + sizeof(hosted_osi_funcs_t) == 284, + "hosted_osi_funcs_t is not the 284-byte layout. Either the force-include of " + "port_esp_hosted_host_config.h was lost (short layout, 268), or ESP-Hosted changed the table. " + "Compare against the struct in src/net/port.zig before touching this number."); + +/* Two offsets, chosen because they sit on either side of the mempool block: the first entry after + * it, and one near the end. If the block appears or disappears, both move. */ +_Static_assert( + offsetof(hosted_osi_funcs_t, _h_config_gpio) == 148, + "_h_config_gpio moved. It is the first entry after the #ifdef H_USE_MEMPOOL block, so this is " + "what the short layout breaks first: 132 instead of 148."); + +_Static_assert( + offsetof(hosted_osi_funcs_t, _h_event_post) == 280, + "_h_event_post moved. Together with the _h_config_gpio assertion this pins both ends of the " + "table."); + +/* Field count, checked through the size. src/net/port.zig asserts its Zig struct has 71 fields; + * every entry is a pointer, so 71 * 4 must be the size on this 32-bit target. A size check alone + * would not catch a field deleted in one place and duplicated in another - the length survives and + * the two sides silently disagree about which pointer is which. */ +_Static_assert( + sizeof(hosted_osi_funcs_t) == 71 * sizeof(void (*)(void)), + "hosted_osi_funcs_t is not 71 function pointers. Compare field by field against the struct in " + "src/net/port.zig - a count mismatch means one side has an entry the other does not, and every " + "entry after it calls the wrong function."); + +const int esp_hosted_abi_assertions_hold = 1; diff --git a/src/net/hosted/include_dirs.txt b/src/net/hosted/include_dirs.txt new file mode 100644 index 0000000..75ee2d3 --- /dev/null +++ b/src/net/hosted/include_dirs.txt @@ -0,0 +1,171 @@ +# Include directories ESP-Hosted's C needs, in order. +# +# Provenance: extracted from the -I flags ESP-IDF v6.0.2 used to compile +# host/drivers/transport/transport_drv.c in 02-esp32p4-m3-radio/build/compile_commands.json - +# the build that worked on this board. Order is IDF's and matters: esp_wifi_remote's +# idf_v6.0/include/injected must precede components/esp_wifi/include, because a host with no +# radio needs the injected Wi-Fi types rather than the real ones. +# +# IDF/ -> relative to the ESP-IDF checkout +# MC/ -> relative to 02-esp32p4-m3-radio (the managed_components tree) +# +# The one path dropped from IDF's list is build/config, the generated Kconfig header. This +# project supplies its own copy as src/net/hosted/sdkconfig.h. +MC/managed_components/espressif__esp_hosted/host +MC/managed_components/espressif__esp_hosted/host/api/include +MC/managed_components/espressif__esp_hosted/host/drivers/transport +MC/managed_components/espressif__esp_hosted/host/drivers/transport/spi +MC/managed_components/espressif__esp_hosted/host/drivers/transport/sdio +MC/managed_components/espressif__esp_hosted/host/drivers/serial +MC/managed_components/espressif__esp_hosted/host/utils +MC/managed_components/espressif__esp_hosted/host/api/priv +MC/managed_components/espressif__esp_hosted/host/drivers/rpc/core +MC/managed_components/espressif__esp_hosted/host/drivers/rpc/slaveif +MC/managed_components/espressif__esp_hosted/host/drivers/rpc/wrap +MC/managed_components/espressif__esp_hosted/host/drivers/virtual_serial_if +MC/managed_components/espressif__esp_hosted/common +MC/managed_components/espressif__esp_hosted/common/log +MC/managed_components/espressif__esp_hosted/common/rpc +MC/managed_components/espressif__esp_hosted/common/transport +MC/managed_components/espressif__esp_hosted/common/protobuf-c +MC/managed_components/espressif__esp_hosted/common/proto +MC/managed_components/espressif__esp_hosted/common/mempool/include +MC/managed_components/espressif__esp_hosted/common/utils +MC/managed_components/espressif__esp_hosted/host/drivers/bt +MC/managed_components/espressif__esp_hosted/host/drivers/power_save +MC/managed_components/espressif__esp_hosted/host/port/esp/freertos/include +IDF/components/esp_libc/platform_include +IDF/components/freertos/config/include +IDF/components/freertos/config/include/freertos +IDF/components/freertos/config/riscv/include +IDF/components/freertos/FreeRTOS-Kernel/include +IDF/components/freertos/FreeRTOS-Kernel/portable/riscv/include +IDF/components/freertos/FreeRTOS-Kernel/portable/riscv/include/freertos +IDF/components/freertos/esp_additions/include +IDF/components/esp_hw_support/include +IDF/components/esp_hw_support/include/soc +IDF/components/esp_hw_support/ldo/include +IDF/components/esp_hw_support/debug_probe/include +IDF/components/esp_hw_support/etm/include +IDF/components/esp_hw_support/mspi_timing_tuning/include +IDF/components/esp_hw_support/mspi_timing_tuning/tuning_scheme_impl/include +IDF/components/esp_hw_support/power_supply/include +IDF/components/esp_hw_support/modem/include +IDF/components/esp_hw_support/port/esp32p4/. +IDF/components/esp_hw_support/port/esp32p4/include +IDF/components/esp_hw_support/port/esp32p4/private_include +IDF/components/esp_hw_support/mspi_timing_tuning/port/esp32p4/. +IDF/components/heap/include +IDF/components/heap/tlsf +IDF/components/log/include +IDF/components/soc/include +IDF/components/soc/esp32p4 +IDF/components/soc/esp32p4/include +IDF/components/soc/esp32p4/register/hw_ver1 +IDF/components/hal/platform_port/include +IDF/components/hal/esp32p4/include +IDF/components/hal/include +IDF/components/esp_rom/include +IDF/components/esp_rom/esp32p4/include +IDF/components/esp_rom/esp32p4/include/esp32p4 +IDF/components/esp_rom/esp32p4 +IDF/components/esp_common/include +IDF/components/esp_system/include +IDF/components/esp_system/port/soc +IDF/components/esp_system/port/include/riscv +IDF/components/esp_system/port/include/private +IDF/components/esp_stdio/include +IDF/components/riscv/include +IDF/components/esp_hal_gpio/include +IDF/components/esp_hal_gpio/esp32p4/include +IDF/components/esp_hal_usb/include +IDF/components/esp_hal_usb/esp32p4/include +IDF/components/esp_hal_pmu/include +IDF/components/esp_hal_pmu/esp32p4/include +IDF/components/esp_hal_ana_conv/include +IDF/components/esp_hal_ana_conv/esp32p4/include +IDF/components/esp_hal_dma/include +IDF/components/esp_hal_dma/esp32p4/include +IDF/components/lwip/include +IDF/components/lwip/include/apps +IDF/components/lwip/lwip/src/include +IDF/components/lwip/port/include +IDF/components/lwip/port/freertos/include +IDF/components/lwip/port/esp32xx/include +IDF/components/lwip/port/esp32xx/include/arch +IDF/components/lwip/port/esp32xx/include/sys +IDF/components/esp_driver_sdmmc/include +IDF/components/esp_driver_sdmmc/legacy/include +IDF/components/esp_driver_sd_intf/include +IDF/components/sdmmc/include +IDF/components/esp_blockdev/include +IDF/components/esp_hal_sd/include +IDF/components/esp_hal_sd/esp32p4/include +IDF/components/esp_driver_spi/include +IDF/components/esp_pm/include +IDF/components/esp_hal_gpspi/include +IDF/components/esp_hal_gpspi/esp32p4/include +IDF/components/esp_driver_dma/include +IDF/components/esp_driver_uart/include +IDF/components/esp_hal_uart/include +IDF/components/esp_hal_uart/esp32p4/include +IDF/components/vfs/include +IDF/components/esp_driver_gpio/include +IDF/components/esp_event/include +IDF/components/esp_netif/include +IDF/components/esp_timer/include +IDF/components/driver/i2c/include +IDF/components/driver/touch_sensor/include +IDF/components/driver/twai/include +IDF/components/esp_hal_i2c/esp32p4/include +IDF/components/esp_hal_i2c/include +IDF/components/esp_hal_twai/include +IDF/components/esp_hal_twai/esp32p4/include +IDF/components/esp_hal_touch_sens/esp32p4/include +IDF/components/esp_hal_touch_sens/include +MC/managed_components/espressif__esp_wifi_remote/idf_v6.0/include/injected +IDF/components/esp_wifi/include +IDF/components/esp_wifi/wifi_apps/nan_app/include +MC/managed_components/espressif__esp_wifi_remote/include +MC/managed_components/espressif__esp_wifi_remote/idf_v6.0/include +IDF/components/bt/common/osi/include +IDF/components/bt/common/api/include/api +IDF/components/bt/common/btc/profile/esp/blufi/include +IDF/components/bt/common/btc/profile/esp/include +IDF/components/bt/common/hci_log/include +IDF/components/bt/common/ble_log/include +IDF/components/bt/common/ble_log/deprecated/include +IDF/components/bt/common/tinycrypt/include +IDF/components/bt/common/tinycrypt/port +IDF/components/bt/host/nimble/nimble/nimble/host/include +IDF/components/bt/host/nimble/nimble/nimble/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/ans/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/bas/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/dis/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/gap/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/gatt/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/hr/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/htp/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/ias/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/ipss/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/lls/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/prox/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/cts/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/tps/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/hid/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/sps/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/cte/include +IDF/components/bt/host/nimble/nimble/nimble/host/util/include +IDF/components/bt/host/nimble/nimble/nimble/host/store/ram/include +IDF/components/bt/host/nimble/nimble/nimble/host/store/config/include +IDF/components/bt/host/nimble/nimble/nimble/host/services/ras/include +IDF/components/bt/host/nimble/nimble/porting/nimble/include +IDF/components/bt/host/nimble/port/include +IDF/components/bt/host/nimble/nimble/nimble/transport/include +IDF/components/bt/host/nimble/nimble/nimble/transport/common/hci_h4/include +IDF/components/bt/porting/include +IDF/components/bt/host/nimble/nimble/porting/npl/freertos/include +IDF/components/esp_http_client/include +IDF/components/console +IDF/components/wpa_supplicant/esp_supplicant/include +IDF/components/esp_driver_usb_serial_jtag/include diff --git a/src/net/hosted/pin_assert.c b/src/net/hosted/pin_assert.c new file mode 100644 index 0000000..8704c7b --- /dev/null +++ b/src/net/hosted/pin_assert.c @@ -0,0 +1,56 @@ +/* + * The board, asserted at compile time. + * + * ESP-Hosted derives its SDIO pin map, bus width, clock and the C6 reset pin from Kconfig, and this + * project checks in that Kconfig verbatim as src/net/hosted/sdkconfig.h. That makes the wiring a + * build input rather than something written in Zig - which is the right choice, because the real + * `struct esp_hosted_sdio_config` interleaves `gpio_pin_t {void *port; int pin;}` pairs and + * transcribing it into Zig invites a silent wrong-pin bug. + * + * The cost of that choice is that the wiring is now several files away from the board. This file + * pays it back: every value is asserted against what was measured on the die. If a Kconfig symbol + * ever drifts, the build fails naming the pin, instead of the C6 quietly never answering - which + * is the same symptom as a dead radio, a wrong clock, or a held reset, and takes an afternoon to + * tell apart. + * + * Measurements: Guition JC-ESP32P4-M3-DEV schematic sheet 5 (schematics/5_ESP32-C6.png in the + * unofficial board repo), confirmed against the boot log of the ESP-IDF build in + * 02-esp32p4-m3-radio that reached esp_hosted transport state "active" on this die. + */ + +#include "port_esp_hosted_host_config.h" + +/* SDIO slot and bus. Slot 1 is the only one routed to the C6 on this board. */ +_Static_assert(H_SDMMC_HOST_SLOT == 1, "SDIO slot: board routes the C6 to slot 1"); +_Static_assert(H_SDIO_BUS_WIDTH == 4, "SDIO bus width: all four data lines are wired"); +_Static_assert(H_SDIO_CLOCK_FREQ_KHZ == 40000, "SDIO clock: 40 MHz was measured working"); + +/* Pin map. D1 doubles as the slave interrupt line, which is why it must be a real data pin and + * not left unconfigured. */ +_Static_assert(H_SDIO_PIN_CLK == 18, "SDIO CLK is GPIO18"); +_Static_assert(H_SDIO_PIN_CMD == 19, "SDIO CMD is GPIO19"); +_Static_assert(H_SDIO_PIN_D0 == 14, "SDIO D0 is GPIO14"); +_Static_assert(H_SDIO_PIN_D1 == 15, "SDIO D1 is GPIO15, and doubles as the slave interrupt"); +_Static_assert(H_SDIO_PIN_D2 == 16, "SDIO D2 is GPIO16"); +_Static_assert(H_SDIO_PIN_D3 == 17, "SDIO D3 is GPIO17"); + +/* The C6 reset. Active low with an external pull-up: it must be RELEASED, never driven high. + * Driving it the other way holds the radio in reset forever while looking like a config detail. */ +_Static_assert(H_GPIO_PIN_RESET == 54, "C6 reset is GPIO54"); + +/* The coprocessor. H_SLAVE_TARGET_ESP32C6 is what + * host/port/esp/freertos/include/port_esp_hosted_host_config.h:65-95 derives from + * CONFIG_ESP_HOSTED_CP_TARGET_ESP32C6, and it gates wire-format decisions further up. */ +#ifndef H_SLAVE_TARGET_ESP32C6 +#error "Slave target is not ESP32-C6. The coprocessor on this board is an ESP32-C6-MINI." +#endif + +/* H_USE_MEMPOOL must be DEFINED - its value is a real choice (see sdkconfig.h override 3), but + * whether the name exists at all is what decides the length of hosted_osi_funcs_t, because the + * struct guards four members with #ifdef. abi_assert.c checks the resulting size directly. */ +#ifndef H_USE_MEMPOOL +#error "H_USE_MEMPOOL is not defined: the force-include of port_esp_hosted_host_config.h is missing." +#endif + +/* A definition, so the translation unit is not empty. */ +const int esp_hosted_pin_assertions_hold = 1; diff --git a/src/net/hosted/sdkconfig.h b/src/net/hosted/sdkconfig.h new file mode 100644 index 0000000..63e77e9 --- /dev/null +++ b/src/net/hosted/sdkconfig.h @@ -0,0 +1,139 @@ +/* + * The Kconfig surface ESP-Hosted's C compiles against in this project. + * + * Two parts, deliberately separated: + * + * sdkconfig_idf.h ESP-IDF v6.0.2's generated header, verbatim, from the build in + * 02-esp32p4-m3-radio that reached transport state "active" on this die. + * Unedited, so its provenance is checkable. + * + * this file that header, plus a short list of overrides. Each one states what it changes + * and why, so the delta from the proven configuration is reviewable rather than + * buried in 1,400 generated lines. + * + * The generated header is used rather than a hand-picked subset because ESP-Hosted's headers derive + * struct layouts and the slave target from these symbols: CONFIG_ESP_HOSTED_CP_TARGET_ESP32C6 is + * what defines H_SLAVE_TARGET_ESP32C6, and CONFIG_ESP_HOSTED_USE_MEMPOOL is what decides the length + * of hosted_osi_funcs_t (see abi_assert.c). Hand-picking would be a second, unproven configuration. + * + * Nothing here starts a FreeRTOS kernel or an IDF component. The CONFIG_FREERTOS_* values only + * shape type declarations; src/net/port.zig and src/io/p4.zig supply the runtime. + */ + +#pragma once + +#include "sdkconfig_idf.h" + +/* ------------------------------------------------------------------------------------------------ + * Override 1: SDIO queue depths, 20 -> 4 each. + * + * IDF's build ran with 20 TX and 20 RX descriptors. That is a reasonable number when the heap is + * PSRAM-backed; here it is not. Forty in-flight buffers at MAX_SDIO_BUFFER_SIZE (1536 B) reserve + * ~60 KB before a single task stack exists, and this image has ~128 KB of L2MEM in total with + * nothing initialising the 32 MB of PSRAM. + * + * Four each was the first attempt and it was measured wrong. Once the board associated, the AP's + * ordinary broadcast traffic filled a four-deep queue immediately: the console filled with + * "task still writing Rx data to queue!", the receive counter froze at 8 frames, and the board + * stopped answering ARP - so it took a DHCP lease and then went silent, which looked like a bug in + * the IP stack rather than a queue two sizes too small. + * + * Sixteen each was then too many, for the reason that makes this setting awkward: with the mempool + * off (override 3) every frame is a fresh `_h_malloc_align(MAX_TRANSPORT_BUFFER_SIZE, 64)` from our + * heap, so the depths bound peak heap demand at (tx + rx) x 1536 bytes. At sixteen each that is + * ~49 KB of a 56 KB heap, and the board duly ran out: "mempool OOM start (RX)" at 11 s, then + * "STA TX: mempool_alloc failed, dropping pkt", after which nothing moved in either direction. + * + * Eight each: ~24.5 KB peak, against a heap sized well above it in examples/http.zig. Deep enough + * that ordinary broadcast traffic does not fill the queue between two `tick`s, shallow enough that a + * burst cannot exhaust the heap and stop the transmit path as collateral damage. That second + * property is the one worth protecting: a receive queue that overflows drops a frame, but a heap + * that empties takes the whole radio down. + * ---------------------------------------------------------------------------------------------- */ +#undef CONFIG_ESP_HOSTED_SDIO_TX_Q_SIZE +#define CONFIG_ESP_HOSTED_SDIO_TX_Q_SIZE 8 +#undef CONFIG_ESP_HOSTED_SDIO_RX_Q_SIZE +#define CONFIG_ESP_HOSTED_SDIO_RX_Q_SIZE 8 + +/* These two are aliases the transport reads; they must follow the values above rather than the + * originals, or the queues and the descriptors disagree about their own depth. */ +#undef CONFIG_ESP_SDIO_TX_Q_SIZE +#define CONFIG_ESP_SDIO_TX_Q_SIZE CONFIG_ESP_HOSTED_SDIO_TX_Q_SIZE +#undef CONFIG_ESP_SDIO_RX_Q_SIZE +#define CONFIG_ESP_SDIO_RX_Q_SIZE CONFIG_ESP_HOSTED_SDIO_RX_Q_SIZE + +/* ------------------------------------------------------------------------------------------------ + * Override 2: Bluetooth off. + * + * The IDF build this configuration came from used BLE through the C6, so it enabled NimBLE and the + * VHCI transport. This project does not do Bluetooth, and leaving it on is not free: ESP-Hosted's + * transport calls hci_drv_init() unconditionally (transport_drv.c:126), and with NimBLE enabled that + * pulls in the real vhci_drv.c and the whole NimBLE host - ble_transport_*, os_mbuf_*, + * ble_hs_mbuf_to_flat - which is another stack this image has no reason to carry. + * + * With these off, ESP-Hosted's own host/drivers/bt/hci_stub_drv.c compiles to a no-op hci_drv_init + * and a drop-everything hci_rx_handler. That file is in the source list in build.zig, which is why + * this is a configuration change rather than a Zig stub: the C already ships the right answer for a + * host without Bluetooth, and using it keeps one fewer thing for us to get wrong. + * + * The C6 still reports HCI capability in its capability byte (0x0d on this board). That is the + * coprocessor saying what it can do, not a request; declining is the host's decision. + * ---------------------------------------------------------------------------------------------- */ +#undef CONFIG_ESP_HOSTED_ENABLE_BT_NIMBLE +#undef CONFIG_ESP_HOSTED_NIMBLE_HCI_VHCI +#undef CONFIG_ESP_HOSTED_ENABLE_BT_BLUEDROID +#undef CONFIG_ESP_HOSTED_BLUEDROID_HCI_VHCI +#undef CONFIG_BT_ENABLED +#undef CONFIG_BT_NIMBLE_ENABLED + +/* ------------------------------------------------------------------------------------------------ + * Override 3: mempool off. + * + * ESP-Hosted's mempool recycles fixed-size packet buffers instead of going to malloc each time. It + * needs a backend, supplied by `os_mempool_get_ops()`, and on ESP-IDF that comes from FreeRTOS's own + * pool implementation. This image has no FreeRTOS, and mempool.c treats a null ops table as a hard + * failure rather than a fallback (mempool.c:63-65, "hosted mempool init failed: no mempool ops") - + * which is what the second run of examples/radio.zig printed. + * + * The choice is to write a pool backend or to switch the optimisation off. Off, for now: the + * allocator behind _h_malloc (src/net/heap.zig) is a coalescing free list over a static buffer, so + * the same-size churn mempool exists to avoid is already cheap and cannot fragment the way a + * general-purpose heap would. If profiling later says otherwise, the backend is a small job and this + * is the one line to flip back. + * + * Layout note, because this looks dangerous and is not: turning this off does NOT change + * hosted_osi_funcs_t. port_esp_hosted_host_config.h:127-131 defines H_USE_MEMPOOL to 1 or to 0, and + * the struct's four mempool members are guarded by `#ifdef`, which only asks whether the name is + * defined. Both ways the struct is 284 bytes. src/net/hosted/abi_assert.c asserts that directly. + * ---------------------------------------------------------------------------------------------- */ +#undef CONFIG_ESP_HOSTED_USE_MEMPOOL +#define CONFIG_ESP_HOSTED_USE_MEMPOOL 0 + +/* ------------------------------------------------------------------------------------------------ + * Override 4: the ESP-Hosted CLI off. + * + * A console command set for poking the transport at runtime. It needs IDF's `console` component - + * esp_console_cmd_register, a line editor, and a UART driver - none of which exists in this image, + * and none of which this project wants: the serial line here is a log, not a shell. + * + * `H_ESP_HOSTED_CLI_ENABLED` is an `#ifdef` on the *value* of this symbol being defined + * (transport_drv.c:804), so it must be #undef'd rather than defined to 0. + * ---------------------------------------------------------------------------------------------- */ +#undef CONFIG_ESP_HOSTED_CLI_ENABLED + +/* ------------------------------------------------------------------------------------------------ + * Override 5: compile DEBUG-level logging in. + * + * The generated configuration stops at CONFIG_LOG_MAXIMUM_LEVEL 3 (INFO), which compiles ESP_LOGD + * away entirely. That hides exactly the lines needed to tell a stalled receive path apart from a + * silent slave: sdio_drv.c:1190 logs "--- Wait for SDIO intr ---" at DEBUG on every pass of + * sdio_read_task, so its presence or absence answers "is the read task still looping?" directly. + * + * 4, not 5: VERBOSE adds a per-interrupt line that floods a 115200 baud console and changes the + * timing of the thing being measured. + * + * The runtime filter in src/net/hosted_glue.zig is separate and independent - this only decides what + * exists in the image to be filtered. + * ---------------------------------------------------------------------------------------------- */ +#undef CONFIG_LOG_MAXIMUM_LEVEL +#define CONFIG_LOG_MAXIMUM_LEVEL 4 diff --git a/src/net/hosted/sdkconfig_idf.h b/src/net/hosted/sdkconfig_idf.h new file mode 100644 index 0000000..1457dcf --- /dev/null +++ b/src/net/hosted/sdkconfig_idf.h @@ -0,0 +1,1473 @@ +/* + * ESP-IDF v6.0.2's generated Kconfig header, verbatim. + * + * Provenance: 00-projects/0x4200.cafe/02-esp32p4-m3-radio/build/config/sdkconfig.h, produced by the + * IDF build that brought this P4 onto Wi-Fi through the onboard ESP32-C6 over SDIO. Not edited - + * not one line. Deviations belong in sdkconfig.h, which includes this file and then overrides + * individual symbols with a reason attached. + * + * Do not include this directly. Include "sdkconfig.h", which is what ESP-IDF's own headers ask for. + */ +/* + * Automatically generated file. DO NOT EDIT. + * Espressif IoT Development Framework (ESP-IDF) 6.0.2 Configuration Header + */ +#pragma once +#define CONFIG_SOC_ADC_SUPPORTED 1 +#define CONFIG_SOC_ANA_CMPR_SUPPORTED 1 +#define CONFIG_SOC_DEDICATED_GPIO_SUPPORTED 1 +#define CONFIG_SOC_UART_SUPPORTED 1 +#define CONFIG_SOC_GDMA_SUPPORTED 1 +#define CONFIG_SOC_UHCI_SUPPORTED 1 +#define CONFIG_SOC_AHB_GDMA_SUPPORTED 1 +#define CONFIG_SOC_AXI_GDMA_SUPPORTED 1 +#define CONFIG_SOC_DW_GDMA_SUPPORTED 1 +#define CONFIG_SOC_DMA2D_SUPPORTED 1 +#define CONFIG_SOC_GPTIMER_SUPPORTED 1 +#define CONFIG_SOC_PCNT_SUPPORTED 1 +#define CONFIG_SOC_LCDCAM_CAM_SUPPORTED 1 +#define CONFIG_SOC_LCDCAM_I80_LCD_SUPPORTED 1 +#define CONFIG_SOC_LCDCAM_RGB_LCD_SUPPORTED 1 +#define CONFIG_SOC_LCD_I80_SUPPORTED 1 +#define CONFIG_SOC_LCD_RGB_SUPPORTED 1 +#define CONFIG_SOC_MIPI_CSI_SUPPORTED 1 +#define CONFIG_SOC_MIPI_DSI_SUPPORTED 1 +#define CONFIG_SOC_MCPWM_SUPPORTED 1 +#define CONFIG_SOC_TWAI_SUPPORTED 1 +#define CONFIG_SOC_ETM_SUPPORTED 1 +#define CONFIG_SOC_PARLIO_SUPPORTED 1 +#define CONFIG_SOC_PARLIO_LCD_SUPPORTED 1 +#define CONFIG_SOC_ASYNC_MEMCPY_SUPPORTED 1 +#define CONFIG_SOC_EMAC_SUPPORTED 1 +#define CONFIG_SOC_USB_OTG_SUPPORTED 1 +#define CONFIG_SOC_WIRELESS_HOST_SUPPORTED 1 +#define CONFIG_SOC_USB_SERIAL_JTAG_SUPPORTED 1 +#define CONFIG_SOC_TEMP_SENSOR_SUPPORTED 1 +#define CONFIG_SOC_SUPPORTS_SECURE_DL_MODE 1 +#define CONFIG_SOC_ULP_SUPPORTED 1 +#define CONFIG_SOC_LP_CORE_SUPPORTED 1 +#define CONFIG_SOC_EFUSE_KEY_PURPOSE_FIELD 1 +#define CONFIG_SOC_EFUSE_SUPPORTED 1 +#define CONFIG_SOC_RTC_FAST_MEM_SUPPORTED 1 +#define CONFIG_SOC_RTC_MEM_SUPPORTED 1 +#define CONFIG_SOC_RMT_SUPPORTED 1 +#define CONFIG_SOC_I2S_SUPPORTED 1 +#define CONFIG_SOC_SDM_SUPPORTED 1 +#define CONFIG_SOC_GPSPI_SUPPORTED 1 +#define CONFIG_SOC_LEDC_SUPPORTED 1 +#define CONFIG_SOC_ISP_SUPPORTED 1 +#define CONFIG_SOC_I2C_SUPPORTED 1 +#define CONFIG_SOC_SYSTIMER_SUPPORTED 1 +#define CONFIG_SOC_AES_SUPPORTED 1 +#define CONFIG_SOC_MPI_SUPPORTED 1 +#define CONFIG_SOC_SHA_SUPPORTED 1 +#define CONFIG_SOC_HMAC_SUPPORTED 1 +#define CONFIG_SOC_DIG_SIGN_SUPPORTED 1 +#define CONFIG_SOC_ECC_SUPPORTED 1 +#define CONFIG_SOC_ECC_EXTENDED_MODES_SUPPORTED 1 +#define CONFIG_SOC_ECDSA_SUPPORTED 1 +#define CONFIG_SOC_KEY_MANAGER_SUPPORTED 1 +#define CONFIG_SOC_HUK_SUPPORTED 1 +#define CONFIG_SOC_FLASH_ENC_SUPPORTED 1 +#define CONFIG_SOC_SECURE_BOOT_SUPPORTED 1 +#define CONFIG_SOC_BOD_SUPPORTED 1 +#define CONFIG_SOC_VBAT_SUPPORTED 1 +#define CONFIG_SOC_APM_SUPPORTED 1 +#define CONFIG_SOC_PMU_SUPPORTED 1 +#define CONFIG_SOC_PMU_PVT_SUPPORTED 1 +#define CONFIG_SOC_PVT_EN_WITH_SLEEP 1 +#define CONFIG_SOC_PVT_RETENTION_BY_REGDMA 1 +#define CONFIG_SOC_DCDC_SUPPORTED 1 +#define CONFIG_SOC_PAU_SUPPORTED 1 +#define CONFIG_SOC_RTC_TIMER_V2_SUPPORTED 1 +#define CONFIG_SOC_ULP_LP_UART_SUPPORTED 1 +#define CONFIG_SOC_LP_GPIO_MATRIX_SUPPORTED 1 +#define CONFIG_SOC_LP_PERIPHERALS_SUPPORTED 1 +#define CONFIG_SOC_LP_I2C_SUPPORTED 1 +#define CONFIG_SOC_LP_I2S_SUPPORTED 1 +#define CONFIG_SOC_LP_SPI_SUPPORTED 1 +#define CONFIG_SOC_LP_ADC_SUPPORTED 1 +#define CONFIG_SOC_LP_VAD_SUPPORTED 1 +#define CONFIG_SOC_LP_MAILBOX_SUPPORTED 1 +#define CONFIG_SOC_SPIRAM_SUPPORTED 1 +#define CONFIG_SOC_PSRAM_DMA_CAPABLE 1 +#define CONFIG_SOC_SDMMC_HOST_SUPPORTED 1 +#define CONFIG_SOC_CLK_TREE_SUPPORTED 1 +#define CONFIG_SOC_ASSIST_DEBUG_SUPPORTED 1 +#define CONFIG_SOC_DEBUG_PROBE_SUPPORTED 1 +#define CONFIG_SOC_WDT_SUPPORTED 1 +#define CONFIG_SOC_SPI_FLASH_SUPPORTED 1 +#define CONFIG_SOC_TOUCH_SENSOR_SUPPORTED 1 +#define CONFIG_SOC_RNG_SUPPORTED 1 +#define CONFIG_SOC_GP_LDO_SUPPORTED 1 +#define CONFIG_SOC_PPA_SUPPORTED 1 +#define CONFIG_SOC_LIGHT_SLEEP_SUPPORTED 1 +#define CONFIG_SOC_DEEP_SLEEP_SUPPORTED 1 +#define CONFIG_SOC_PM_SUPPORTED 1 +#define CONFIG_SOC_BITSCRAMBLER_SUPPORTED 1 +#define CONFIG_SOC_SIMD_INSTRUCTION_SUPPORTED 1 +#define CONFIG_SOC_I3C_MASTER_SUPPORTED 1 +#define CONFIG_SOC_XTAL_SUPPORT_40M 1 +#define CONFIG_SOC_AES_SUPPORT_DMA 1 +#define CONFIG_SOC_AES_SUPPORT_GCM 1 +#define CONFIG_SOC_AES_GDMA 1 +#define CONFIG_SOC_AES_SUPPORT_AES_128 1 +#define CONFIG_SOC_AES_SUPPORT_AES_256 1 +#define CONFIG_SOC_AES_SUPPORT_PSEUDO_ROUND_FUNCTION 1 +#define CONFIG_SOC_ADC_RTC_CTRL_SUPPORTED 1 +#define CONFIG_SOC_ADC_DIG_CTRL_SUPPORTED 1 +#define CONFIG_SOC_ADC_DMA_SUPPORTED 1 +#define CONFIG_SOC_ADC_PERIPH_NUM 2 +#define CONFIG_SOC_ADC_MAX_CHANNEL_NUM 8 +#define CONFIG_SOC_ADC_ATTEN_NUM 4 +#define CONFIG_SOC_ADC_DIGI_CONTROLLER_NUM 2 +#define CONFIG_SOC_ADC_PATT_LEN_MAX 16 +#define CONFIG_SOC_ADC_DIGI_MAX_BITWIDTH 12 +#define CONFIG_SOC_ADC_DIGI_MIN_BITWIDTH 12 +#define CONFIG_SOC_ADC_DIGI_IIR_FILTER_NUM 2 +#define CONFIG_SOC_ADC_DIGI_MONITOR_NUM 2 +#define CONFIG_SOC_ADC_DIGI_RESULT_BYTES 4 +#define CONFIG_SOC_ADC_DIGI_DATA_BYTES_PER_CONV 4 +#define CONFIG_SOC_ADC_SAMPLE_FREQ_THRES_HIGH 83333 +#define CONFIG_SOC_ADC_SAMPLE_FREQ_THRES_LOW 611 +#define CONFIG_SOC_ADC_RTC_MIN_BITWIDTH 12 +#define CONFIG_SOC_ADC_RTC_MAX_BITWIDTH 12 +#define CONFIG_SOC_ADC_CALIBRATION_V1_SUPPORTED 1 +#define CONFIG_SOC_ADC_SELF_HW_CALI_SUPPORTED 1 +#define CONFIG_SOC_ADC_CALIB_CHAN_COMPENS_SUPPORTED 1 +#define CONFIG_SOC_ADC_SHARED_POWER 1 +#define CONFIG_SOC_BROWNOUT_RESET_SUPPORTED 1 +#define CONFIG_SOC_SHARED_IDCACHE_SUPPORTED 1 +#define CONFIG_SOC_CACHE_WRITEBACK_SUPPORTED 1 +#define CONFIG_SOC_CACHE_FREEZE_SUPPORTED 1 +#define CONFIG_SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE 1 +#define CONFIG_SOC_CPU_CORES_NUM 2 +#define CONFIG_SOC_CPU_INTR_NUM 32 +#define CONFIG_SOC_CPU_HAS_FLEXIBLE_INTC 1 +#define CONFIG_SOC_INT_CLIC_SUPPORTED 1 +#define CONFIG_SOC_INT_HW_NESTED_SUPPORTED 1 +#define CONFIG_SOC_BRANCH_PREDICTOR_SUPPORTED 1 +#define CONFIG_SOC_CPU_COPROC_NUM 3 +#define CONFIG_SOC_CPU_HAS_FPU 1 +#define CONFIG_SOC_CPU_HAS_FPU_EXT_ILL_BUG 1 +#define CONFIG_SOC_CPU_HAS_HWLOOP 1 +#define CONFIG_SOC_CPU_HAS_HWLOOP_STATE_BUG 1 +#define CONFIG_SOC_CPU_HAS_PIE 1 +#define CONFIG_SOC_HP_CPU_HAS_MULTIPLE_CORES 1 +#define CONFIG_SOC_CPU_BREAKPOINTS_NUM 3 +#define CONFIG_SOC_CPU_WATCHPOINTS_NUM 3 +#define CONFIG_SOC_CPU_WATCHPOINT_MAX_REGION_SIZE 0x100 +#define CONFIG_SOC_CPU_HAS_PMA 1 +#define CONFIG_SOC_CPU_IDRAM_SPLIT_USING_PMP 1 +#define CONFIG_SOC_CPU_PMP_REGION_GRANULARITY 128 +#define CONFIG_SOC_CPU_HAS_LOCKUP_RESET 1 +#define CONFIG_SOC_CPU_HAS_ZC_EXTENSIONS 1 +#define CONFIG_SOC_CPU_ZCMP_WORKAROUND 1 +#define CONFIG_SOC_CPU_ZCMP_PUSH_REVERSED 1 +#define CONFIG_SOC_CPU_ZCMP_POPRET_ISSUE 1 +#define CONFIG_SOC_SIMD_PREFERRED_DATA_ALIGNMENT 16 +#define CONFIG_SOC_DS_SIGNATURE_MAX_BIT_LEN 4096 +#define CONFIG_SOC_DS_KEY_PARAM_MD_IV_LENGTH 16 +#define CONFIG_SOC_DS_KEY_CHECK_MAX_WAIT_US 1100 +#define CONFIG_SOC_DMA_CAN_ACCESS_FLASH 1 +#define CONFIG_SOC_AHB_GDMA_VERSION 2 +#define CONFIG_SOC_GDMA_SUPPORT_CRC 1 +#define CONFIG_SOC_GDMA_SUPPORT_ETM 1 +#define CONFIG_SOC_GDMA_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_GDMA_EXT_MEM_ENC_ALIGNMENT 16 +#define CONFIG_SOC_GPIO_PORT 1 +#define CONFIG_SOC_GPIO_PIN_COUNT 55 +#define CONFIG_SOC_GPIO_SUPPORT_PIN_GLITCH_FILTER 1 +#define CONFIG_SOC_GPIO_FLEX_GLITCH_FILTER_NUM 8 +#define CONFIG_SOC_GPIO_SUPPORT_PIN_HYS_FILTER 1 +#define CONFIG_SOC_GPIO_SUPPORT_ETM 1 +#define CONFIG_SOC_GPIO_SUPPORT_HP_PERIPH_PD_SLEEP_WAKEUP 1 +#define CONFIG_SOC_LP_IO_HAS_INDEPENDENT_WAKEUP_SOURCE 1 +#define CONFIG_SOC_LP_IO_CLOCK_IS_INDEPENDENT 1 +#define CONFIG_SOC_GPIO_VALID_GPIO_MASK 0x007FFFFFFFFFFFFF +#define CONFIG_SOC_GPIO_IN_RANGE_MAX 54 +#define CONFIG_SOC_GPIO_OUT_RANGE_MAX 54 +#define CONFIG_SOC_GPIO_HP_PERIPH_PD_SLEEP_WAKEABLE_MASK 0 +#define CONFIG_SOC_GPIO_HP_PERIPH_PD_SLEEP_WAKEABLE_PIN_CNT 16 +#define CONFIG_SOC_GPIO_VALID_DIGITAL_IO_PAD_MASK 0x007FFFFFFFFF0000 +#define CONFIG_SOC_GPIO_SUPPORT_FORCE_HOLD 1 +#define CONFIG_SOC_GPIO_SUPPORT_HOLD_SINGLE_IO_IN_DSLP 1 +#define CONFIG_SOC_GPIO_CLOCKOUT_BY_GPIO_MATRIX 1 +#define CONFIG_SOC_GPIO_CLOCKOUT_CHANNEL_NUM 2 +#define CONFIG_SOC_CLOCKOUT_SUPPORT_CHANNEL_DIVIDER 1 +#define CONFIG_SOC_DEBUG_PROBE_NUM_UNIT 1 +#define CONFIG_SOC_DEBUG_PROBE_MAX_OUTPUT_WIDTH 16 +#define CONFIG_SOC_RTCIO_PIN_COUNT 16 +#define CONFIG_SOC_RTCIO_INPUT_OUTPUT_SUPPORTED 1 +#define CONFIG_SOC_RTCIO_HOLD_SUPPORTED 1 +#define CONFIG_SOC_RTCIO_WAKE_SUPPORTED 1 +#define CONFIG_SOC_SDM_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_ETM_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_ANA_CMPR_NUM 2 +#define CONFIG_SOC_ANA_CMPR_CAN_DISTINGUISH_EDGE 1 +#define CONFIG_SOC_ANA_CMPR_SUPPORT_ETM 1 +#define CONFIG_SOC_I2C_NUM 3 +#define CONFIG_SOC_HP_I2C_NUM 2 +#define CONFIG_SOC_LP_I2C_NUM 1 +#define CONFIG_SOC_I2C_SUPPORT_XTAL 1 +#define CONFIG_SOC_I2C_SUPPORT_RTC 1 +#define CONFIG_SOC_I2C_SUPPORT_10BIT_ADDR 1 +#define CONFIG_SOC_I2C_SUPPORT_SLAVE 1 +#define CONFIG_SOC_I2C_SLAVE_SUPPORT_BROADCAST 1 +#define CONFIG_SOC_I2C_SLAVE_CAN_GET_STRETCH_CAUSE 1 +#define CONFIG_SOC_I2C_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_I2S_HW_VERSION_2 1 +#define CONFIG_SOC_I2S_SUPPORTS_ETM 1 +#define CONFIG_SOC_I2S_SUPPORTS_APLL 1 +#define CONFIG_SOC_I2S_SUPPORTS_PCM 1 +#define CONFIG_SOC_I2S_SUPPORTS_PDM 1 +#define CONFIG_SOC_I2S_SUPPORTS_PDM_TX 1 +#define CONFIG_SOC_I2S_SUPPORTS_PCM2PDM 1 +#define CONFIG_SOC_I2S_SUPPORTS_PDM_RX 1 +#define CONFIG_SOC_I2S_SUPPORTS_PDM2PCM 1 +#define CONFIG_SOC_I2S_SUPPORTS_PDM_RX_HP_FILTER 1 +#define CONFIG_SOC_I2S_SUPPORTS_TX_SYNC_CNT 1 +#define CONFIG_SOC_I2S_SUPPORTS_TDM 1 +#define CONFIG_SOC_I2S_PDM_MAX_TX_LINES 2 +#define CONFIG_SOC_I2S_PDM_MAX_RX_LINES 4 +#define CONFIG_SOC_LP_I2S_NUM 1 +#define CONFIG_SOC_ISP_BF_SUPPORTED 1 +#define CONFIG_SOC_ISP_BLC_SUPPORTED 1 +#define CONFIG_SOC_ISP_CCM_SUPPORTED 1 +#define CONFIG_SOC_ISP_COLOR_SUPPORTED 1 +#define CONFIG_SOC_ISP_CROP_SUPPORTED 1 +#define CONFIG_SOC_ISP_DEMOSAIC_SUPPORTED 1 +#define CONFIG_SOC_ISP_DVP_SUPPORTED 1 +#define CONFIG_SOC_ISP_LSC_SUPPORTED 1 +#define CONFIG_SOC_ISP_SHARPEN_SUPPORTED 1 +#define CONFIG_SOC_ISP_WBG_SUPPORTED 1 +#define CONFIG_SOC_ISP_SHARE_CSI_BRG 1 +#define CONFIG_SOC_ISP_AE_BLOCK_X_NUMS 5 +#define CONFIG_SOC_ISP_AE_BLOCK_Y_NUMS 5 +#define CONFIG_SOC_ISP_AF_WINDOW_NUMS 3 +#define CONFIG_SOC_ISP_AWB_WINDOW_X_NUMS 5 +#define CONFIG_SOC_ISP_AWB_WINDOW_Y_NUMS 5 +#define CONFIG_SOC_ISP_BF_TEMPLATE_X_NUMS 3 +#define CONFIG_SOC_ISP_BF_TEMPLATE_Y_NUMS 3 +#define CONFIG_SOC_ISP_CCM_DIMENSION 3 +#define CONFIG_SOC_ISP_DEMOSAIC_GRAD_RATIO_INT_BITS 2 +#define CONFIG_SOC_ISP_DEMOSAIC_GRAD_RATIO_DEC_BITS 4 +#define CONFIG_SOC_ISP_DEMOSAIC_GRAD_RATIO_RES_BITS 26 +#define CONFIG_SOC_ISP_SHARPEN_TEMPLATE_X_NUMS 3 +#define CONFIG_SOC_ISP_SHARPEN_TEMPLATE_Y_NUMS 3 +#define CONFIG_SOC_ISP_SHARPEN_H_FREQ_COEF_INT_BITS 3 +#define CONFIG_SOC_ISP_SHARPEN_H_FREQ_COEF_DEC_BITS 5 +#define CONFIG_SOC_ISP_SHARPEN_H_FREQ_COEF_RES_BITS 24 +#define CONFIG_SOC_ISP_SHARPEN_M_FREQ_COEF_INT_BITS 3 +#define CONFIG_SOC_ISP_SHARPEN_M_FREQ_COEF_DEC_BITS 5 +#define CONFIG_SOC_ISP_SHARPEN_M_FREQ_COEF_RES_BITS 24 +#define CONFIG_SOC_ISP_HIST_BLOCK_X_NUMS 5 +#define CONFIG_SOC_ISP_HIST_BLOCK_Y_NUMS 5 +#define CONFIG_SOC_ISP_HIST_SEGMENT_NUMS 16 +#define CONFIG_SOC_ISP_HIST_INTERVAL_NUMS 15 +#define CONFIG_SOC_ISP_LSC_GRAD_RATIO_INT_BITS 2 +#define CONFIG_SOC_ISP_LSC_GRAD_RATIO_DEC_BITS 8 +#define CONFIG_SOC_ISP_LSC_GRAD_RATIO_RES_BITS 22 +#define CONFIG_SOC_LEDC_SUPPORT_PLL_DIV_CLOCK 1 +#define CONFIG_SOC_LEDC_SUPPORT_XTAL_CLOCK 1 +#define CONFIG_SOC_LEDC_TIMER_NUM 4 +#define CONFIG_SOC_LEDC_CHANNEL_NUM 8 +#define CONFIG_SOC_LEDC_TIMER_BIT_WIDTH 20 +#define CONFIG_SOC_LEDC_GAMMA_CURVE_FADE_SUPPORTED 1 +#define CONFIG_SOC_LEDC_GAMMA_CURVE_FADE_RANGE_MAX 16 +#define CONFIG_SOC_LEDC_SUPPORT_FADE_STOP 1 +#define CONFIG_SOC_LEDC_FADE_PARAMS_BIT_WIDTH 10 +#define CONFIG_SOC_LEDC_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_LEDC_SUPPORT_ETM 1 +#define CONFIG_SOC_MMU_PERIPH_NUM 2 +#define CONFIG_SOC_MMU_LINEAR_ADDRESS_REGION_NUM 2 +#define CONFIG_SOC_MMU_DI_VADDR_SHARED 1 +#define CONFIG_SOC_MMU_PER_EXT_MEM_TARGET 1 +#define CONFIG_SOC_MPU_MIN_REGION_SIZE 0x20000000 +#define CONFIG_SOC_MPU_REGIONS_MAX_NUM 8 +#define CONFIG_SOC_PCNT_SUPPORT_RUNTIME_THRES_UPDATE 1 +#define CONFIG_SOC_PCNT_SUPPORT_CLEAR_SIGNAL 1 +#define CONFIG_SOC_RMT_MEM_WORDS_PER_CHANNEL 48 +#define CONFIG_SOC_RMT_SUPPORT_RX_PINGPONG 1 +#define CONFIG_SOC_RMT_SUPPORT_TX_LOOP_COUNT 1 +#define CONFIG_SOC_RMT_SUPPORT_TX_LOOP_AUTO_STOP 1 +#define CONFIG_SOC_RMT_SUPPORT_DMA 1 +#define CONFIG_SOC_RMT_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_MCPWM_SWSYNC_CAN_PROPAGATE 1 +#define CONFIG_SOC_MCPWM_SUPPORT_ETM 1 +#define CONFIG_SOC_MCPWM_SUPPORT_EVENT_COMPARATOR 1 +#define CONFIG_SOC_MCPWM_CAPTURE_CLK_FROM_GROUP 1 +#define CONFIG_SOC_MCPWM_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_USB_OTG_PERIPH_NUM 2 +#define CONFIG_SOC_USB_FSLS_PHY_NUM 1 +#define CONFIG_SOC_USB_UTMI_PHY_NUM 1 +#define CONFIG_SOC_USB_UTMI_PHY_NO_POWER_OFF_ISO 1 +#define CONFIG_SOC_PARLIO_TX_UNIT_MAX_DATA_WIDTH 16 +#define CONFIG_SOC_PARLIO_RX_UNIT_MAX_DATA_WIDTH 16 +#define CONFIG_SOC_PARLIO_TX_CLK_SUPPORT_GATING 1 +#define CONFIG_SOC_PARLIO_RX_CLK_SUPPORT_GATING 1 +#define CONFIG_SOC_PARLIO_TX_SUPPORT_LOOP_TRANSMISSION 1 +#define CONFIG_SOC_PARLIO_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_PARLIO_SUPPORT_I80_LCD 1 +#define CONFIG_SOC_MPI_MEM_BLOCKS_NUM 4 +#define CONFIG_SOC_MPI_OPERATIONS_NUM 3 +#define CONFIG_SOC_RSA_MAX_BIT_LEN 4096 +#define CONFIG_SOC_SDMMC_USE_IOMUX 1 +#define CONFIG_SOC_SDMMC_USE_GPIO_MATRIX 1 +#define CONFIG_SOC_SDMMC_NUM_SLOTS 2 +#define CONFIG_SOC_SDMMC_DATA_WIDTH_MAX 8 +#define CONFIG_SOC_SDMMC_DELAY_PHASE_NUM 8 +#define CONFIG_SOC_SDMMC_IO_POWER_EXTERNAL 1 +#define CONFIG_SOC_SDMMC_PSRAM_DMA_CAPABLE 1 +#define CONFIG_SOC_SDMMC_UHS_I_SUPPORTED 1 +#define CONFIG_SOC_SHA_DMA_MAX_BUFFER_SIZE 3968 +#define CONFIG_SOC_SHA_SUPPORT_DMA 1 +#define CONFIG_SOC_SHA_SUPPORT_RESUME 1 +#define CONFIG_SOC_SHA_GDMA 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA1 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA224 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA256 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA384 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA512 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA512_224 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA512_256 1 +#define CONFIG_SOC_SHA_SUPPORT_SHA512_T 1 +#define CONFIG_SOC_ECC_CONSTANT_TIME_POINT_MUL 1 +#define CONFIG_SOC_ECC_SUPPORT_CURVE_P384 1 +#define CONFIG_SOC_ECDSA_SUPPORT_EXPORT_PUBKEY 1 +#define CONFIG_SOC_ECDSA_SUPPORT_DETERMINISTIC_MODE 1 +#define CONFIG_SOC_ECDSA_SUPPORT_HW_DETERMINISTIC_LOOP 1 +#define CONFIG_SOC_ECDSA_USES_MPI 1 +#define CONFIG_SOC_ECDSA_SUPPORT_CURVE_P384 1 +#define CONFIG_SOC_ECDSA_SUPPORT_CURVE_SPECIFIC_KEY_PURPOSES 1 +#define CONFIG_SOC_SPI_PERIPH_NUM 3 +#define CONFIG_SOC_SPI_MAX_CS_NUM 6 +#define CONFIG_SOC_SPI_MAXIMUM_BUFFER_SIZE 64 +#define CONFIG_SOC_SPI_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_SPI_SUPPORT_SLAVE_HD_VER2 1 +#define CONFIG_SOC_SPI_SLAVE_SUPPORT_SEG_TRANS 1 +#define CONFIG_SOC_SPI_SUPPORT_DDRCLK 1 +#define CONFIG_SOC_SPI_SUPPORT_CD_SIG 1 +#define CONFIG_SOC_SPI_SUPPORT_OCT 1 +#define CONFIG_SOC_SPI_SUPPORT_CLK_XTAL 1 +#define CONFIG_SOC_SPI_SUPPORT_CLK_RC_FAST 1 +#define CONFIG_SOC_MSPI_HAS_INDEPENT_IOMUX 1 +#define CONFIG_SOC_MEMSPI_IS_INDEPENDENT 1 +#define CONFIG_SOC_SPI_MAX_PRE_DIVIDER 16 +#define CONFIG_SOC_LP_SPI_MAXIMUM_BUFFER_SIZE 64 +#define CONFIG_SOC_SPIRAM_XIP_SUPPORTED 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_AUTO_WAIT_IDLE 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_AUTO_SUSPEND 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_AUTO_RESUME 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_IDLE_INTR 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_SW_SUSPEND 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_CHECK_SUS 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_TIMING_TUNING 1 +#define CONFIG_SOC_MEMSPI_TIMING_TUNING_BY_DQS 1 +#define CONFIG_SOC_MEMSPI_TIMING_TUNING_BY_FLASH_DELAY 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_CACHE_32BIT_ADDR_MAP 1 +#define CONFIG_SOC_SPI_MEM_SUPPORT_TSUS_TRES_SEPERATE_CTR 1 +#define CONFIG_SOC_SPI_PERIPH_SUPPORT_CONTROL_DUMMY_OUT 1 +#define CONFIG_SOC_SPI_MEM_FLASH_SUPPORT_HPM 1 +#define CONFIG_SOC_MEMSPI_ENCRYPTION_ALIGNMENT 16 +#define CONFIG_SOC_SYSTIMER_COUNTER_NUM 2 +#define CONFIG_SOC_SYSTIMER_ALARM_NUM 3 +#define CONFIG_SOC_SYSTIMER_BIT_WIDTH_LO 32 +#define CONFIG_SOC_SYSTIMER_BIT_WIDTH_HI 20 +#define CONFIG_SOC_SYSTIMER_FIXED_DIVIDER 1 +#define CONFIG_SOC_SYSTIMER_SUPPORT_RC_FAST 1 +#define CONFIG_SOC_SYSTIMER_INT_LEVEL 1 +#define CONFIG_SOC_SYSTIMER_ALARM_MISS_COMPENSATE 1 +#define CONFIG_SOC_SYSTIMER_SUPPORT_ETM 1 +#define CONFIG_SOC_LP_TIMER_BIT_WIDTH_LO 32 +#define CONFIG_SOC_LP_TIMER_BIT_WIDTH_HI 16 +#define CONFIG_SOC_TIMER_SUPPORT_ETM 1 +#define CONFIG_SOC_TIMER_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_MWDT_SUPPORT_XTAL 1 +#define CONFIG_SOC_MWDT_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_TOUCH_SENSOR_VERSION 3 +#define CONFIG_SOC_TOUCH_MIN_CHAN_ID 1 +#define CONFIG_SOC_TOUCH_MAX_CHAN_ID 14 +#define CONFIG_SOC_TOUCH_SUPPORT_SLEEP_WAKEUP 1 +#define CONFIG_SOC_TOUCH_SUPPORT_BENCHMARK 1 +#define CONFIG_SOC_TOUCH_SUPPORT_WATERPROOF 1 +#define CONFIG_SOC_TOUCH_SUPPORT_PROX_SENSING 1 +#define CONFIG_SOC_TOUCH_PROXIMITY_CHANNEL_NUM 3 +#define CONFIG_SOC_TOUCH_SAMPLE_CFG_NUM 3 +#define CONFIG_SOC_TWAI_CONTROLLER_NUM 3 +#define CONFIG_SOC_TWAI_MASK_FILTER_NUM 1 +#define CONFIG_SOC_TWAI_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_EFUSE_DIS_PAD_JTAG 1 +#define CONFIG_SOC_EFUSE_DIS_USB_JTAG 1 +#define CONFIG_SOC_EFUSE_DIS_DIRECT_BOOT 1 +#define CONFIG_SOC_EFUSE_SOFT_DIS_JTAG 1 +#define CONFIG_SOC_EFUSE_DIS_DOWNLOAD_MSPI 1 +#define CONFIG_SOC_EFUSE_ECDSA_KEY 1 +#define CONFIG_SOC_EFUSE_XTS_AES_KEY_128 1 +#define CONFIG_SOC_EFUSE_XTS_AES_KEY_256 1 +#define CONFIG_SOC_EFUSE_ECDSA_KEY_P192 1 +#define CONFIG_SOC_EFUSE_ECDSA_KEY_P384 1 +#define CONFIG_SOC_KEY_MANAGER_SUPPORT_KEY_DEPLOYMENT 1 +#define CONFIG_SOC_KEY_MANAGER_ECDSA_KEY_DEPLOY 1 +#define CONFIG_SOC_KEY_MANAGER_FE_KEY_DEPLOY 1 +#define CONFIG_SOC_KEY_MANAGER_FE_KEY_DEPLOY_XTS_AES_128 1 +#define CONFIG_SOC_KEY_MANAGER_FE_KEY_DEPLOY_XTS_AES_256 1 +#define CONFIG_SOC_KEY_MANAGER_HMAC_KEY_DEPLOY 1 +#define CONFIG_SOC_KEY_MANAGER_DS_KEY_DEPLOY 1 +#define CONFIG_SOC_SECURE_BOOT_V2_RSA 1 +#define CONFIG_SOC_SECURE_BOOT_V2_ECC 1 +#define CONFIG_SOC_EFUSE_SECURE_BOOT_KEY_DIGESTS 3 +#define CONFIG_SOC_EFUSE_REVOKE_BOOT_KEY_DIGESTS 1 +#define CONFIG_SOC_SUPPORT_SECURE_BOOT_REVOKE_KEY 1 +#define CONFIG_SOC_FLASH_ENCRYPTED_XTS_AES_BLOCK_MAX 64 +#define CONFIG_SOC_FLASH_ENCRYPTION_XTS_AES 1 +#define CONFIG_SOC_FLASH_ENCRYPTION_XTS_AES_OPTIONS 1 +#define CONFIG_SOC_FLASH_ENCRYPTION_XTS_AES_128 1 +#define CONFIG_SOC_FLASH_ENCRYPTION_XTS_AES_256 1 +#define CONFIG_SOC_FLASH_ENCRYPTION_XTS_AES_SUPPORT_PSEUDO_ROUND 1 +#define CONFIG_SOC_FLASH_ENCRYPTION_PAGE_CONFIGURABLE 1 +#define CONFIG_SOC_PSRAM_ENCRYPTION_SEPARATE_KEY 1 +#define CONFIG_SOC_PSRAM_ENCRYPTION_PAGE_CONFIGURABLE 1 +#define CONFIG_SOC_RECOVERY_BOOTLOADER_SUPPORTED 1 +#define CONFIG_SOC_UART_NUM 6 +#define CONFIG_SOC_UART_HP_NUM 5 +#define CONFIG_SOC_UART_LP_NUM 1 +#define CONFIG_SOC_UART_FIFO_LEN 128 +#define CONFIG_SOC_LP_UART_FIFO_LEN 16 +#define CONFIG_SOC_UART_BITRATE_MAX 5000000 +#define CONFIG_SOC_UART_SUPPORT_RTC_CLK 1 +#define CONFIG_SOC_UART_SUPPORT_XTAL_CLK 1 +#define CONFIG_SOC_UART_SUPPORT_WAKEUP_INT 1 +#define CONFIG_SOC_UART_HAS_LP_UART 1 +#define CONFIG_SOC_UART_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_UART_WAKEUP_CHARS_SEQ_MAX_LEN 5 +#define CONFIG_SOC_UART_WAKEUP_SUPPORT_ACTIVE_THRESH_MODE 1 +#define CONFIG_SOC_UART_WAKEUP_SUPPORT_FIFO_THRESH_MODE 1 +#define CONFIG_SOC_UART_WAKEUP_SUPPORT_START_BIT_MODE 1 +#define CONFIG_SOC_UART_WAKEUP_SUPPORT_CHAR_SEQ_MODE 1 +#define CONFIG_SOC_LP_I2S_SUPPORT_VAD 1 +#define CONFIG_SOC_UHCI_NUM 1 +#define CONFIG_SOC_COEX_HW_PTI 1 +#define CONFIG_SOC_PHY_DIG_REGS_MEM_SIZE 21 +#define CONFIG_SOC_WIFI_LIGHT_SLEEP_CLK_WIDTH 12 +#define CONFIG_SOC_PM_SUPPORT_EXT1_WAKEUP 1 +#define CONFIG_SOC_PM_SUPPORT_EXT1_WAKEUP_MODE_PER_PIN 1 +#define CONFIG_SOC_PM_EXT1_WAKEUP_BY_PMU 1 +#define CONFIG_SOC_PM_SUPPORT_WIFI_WAKEUP 1 +#define CONFIG_SOC_PM_SUPPORT_TOUCH_SENSOR_WAKEUP 1 +#define CONFIG_SOC_PM_SUPPORT_LP_UART_WAKEUP 1 +#define CONFIG_SOC_PM_SUPPORT_CPU_PD 1 +#define CONFIG_SOC_PM_SUPPORT_XTAL32K_PD 1 +#define CONFIG_SOC_PM_SUPPORT_RC32K_PD 1 +#define CONFIG_SOC_PM_SUPPORT_RC_FAST_PD 1 +#define CONFIG_SOC_PM_SUPPORT_VDDSDIO_PD 1 +#define CONFIG_SOC_PM_SUPPORT_TOP_PD 1 +#define CONFIG_SOC_PM_SUPPORT_CNNT_PD 1 +#define CONFIG_SOC_PM_SUPPORT_RTC_PERIPH_PD 1 +#define CONFIG_SOC_PM_SUPPORT_DEEPSLEEP_CHECK_STUB_ONLY 1 +#define CONFIG_SOC_PM_CPU_RETENTION_BY_SW 1 +#define CONFIG_SOC_PM_FPU_RETENTION_BY_SW 1 +#define CONFIG_SOC_PM_CACHE_RETENTION_BY_PAU 1 +#define CONFIG_SOC_PM_PAU_LINK_NUM 4 +#define CONFIG_SOC_PM_PAU_REGDMA_LINK_MULTI_ADDR 1 +#define CONFIG_SOC_PAU_IN_TOP_DOMAIN 1 +#define CONFIG_SOC_PM_PAU_REGDMA_UPDATE_CACHE_BEFORE_WAIT_COMPARE 1 +#define CONFIG_SOC_SLEEP_SYSTIMER_STALL_WORKAROUND 1 +#define CONFIG_SOC_SLEEP_TGWDT_STOP_WORKAROUND 1 +#define CONFIG_SOC_PM_RETENTION_MODULE_NUM 64 +#define CONFIG_SOC_CLK_RC_FAST_SUPPORT_CALIBRATION 1 +#define CONFIG_SOC_CLK_APLL_SUPPORTED 1 +#define CONFIG_SOC_CLK_MPLL_SUPPORTED 1 +#define CONFIG_SOC_CLK_XTAL32K_SUPPORTED 1 +#define CONFIG_SOC_CLK_RC32K_SUPPORTED 1 +#define CONFIG_SOC_CLK_LP_FAST_SUPPORT_LP_PLL 1 +#define CONFIG_SOC_CLK_LP_FAST_SUPPORT_XTAL 1 +#define CONFIG_SOC_PERIPH_CLK_CTRL_SHARED 1 +#define CONFIG_SOC_TEMPERATURE_SENSOR_INTR_SUPPORT 1 +#define CONFIG_SOC_TSENS_IS_INDEPENDENT_FROM_ADC 1 +#define CONFIG_SOC_TEMPERATURE_SENSOR_SUPPORT_ETM 1 +#define CONFIG_SOC_TEMPERATURE_SENSOR_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_MEM_SPM_SUPPORTED 1 +#define CONFIG_SOC_ASYNCHRONOUS_BUS_ERROR_MODE 1 +#define CONFIG_SOC_EMAC_IEEE1588V2_SUPPORTED 1 +#define CONFIG_SOC_EMAC_USE_MULTI_IO_MUX 1 +#define CONFIG_SOC_EMAC_MII_USE_GPIO_MATRIX 1 +#define CONFIG_SOC_EMAC_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_JPEG_CODEC_SUPPORTED 1 +#define CONFIG_SOC_JPEG_DECODE_SUPPORTED 1 +#define CONFIG_SOC_JPEG_ENCODE_SUPPORTED 1 +#define CONFIG_SOC_H264_ENCODER_SUPPORTED 1 +#define CONFIG_SOC_LCDCAM_CAM_SUPPORT_RGB_YUV_CONV 1 +#define CONFIG_SOC_LCDCAM_LCD_SUPPORT_SLEEP_RETENTION 1 +#define CONFIG_SOC_I3C_MASTER_PERIPH_NUM 1 +#define CONFIG_SOC_I3C_MASTER_ADDRESS_TABLE_NUM 12 +#define CONFIG_SOC_I3C_MASTER_COMMAND_TABLE_NUM 12 +#define CONFIG_SOC_LP_CORE_SUPPORT_ETM 1 +#define CONFIG_SOC_LP_CORE_SUPPORT_LP_ADC 1 +#define CONFIG_SOC_LP_CORE_SUPPORT_STORE_LOAD_EXCEPTIONS 1 +#define CONFIG_IDF_CMAKE 1 +#define CONFIG_IDF_TOOLCHAIN "gcc" +#define CONFIG_IDF_TOOLCHAIN_GCC 1 +#define CONFIG_IDF_TARGET_ARCH_RISCV 1 +#define CONFIG_IDF_TARGET_ARCH "riscv" +#define CONFIG_IDF_TARGET "esp32p4" +#define CONFIG_IDF_INIT_VERSION "6.0.2" +#define CONFIG_IDF_TARGET_ESP32P4 1 +#define CONFIG_IDF_FIRMWARE_CHIP_ID 0x0012 +#define CONFIG_APP_BUILD_TYPE_APP_2NDBOOT 1 +#define CONFIG_APP_BUILD_GENERATE_BINARIES 1 +#define CONFIG_APP_BUILD_BOOTLOADER 1 +#define CONFIG_APP_BUILD_USE_FLASH_SECTIONS 1 +#define CONFIG_BOOTLOADER_COMPILE_TIME_DATE 1 +#define CONFIG_BOOTLOADER_PROJECT_VER 1 +#define CONFIG_BOOTLOADER_OFFSET_IN_FLASH 0x2000 +#define CONFIG_BOOTLOADER_COMPILER_OPTIMIZATION_SIZE 1 +#define CONFIG_BOOTLOADER_LOG_VERSION_1 1 +#define CONFIG_BOOTLOADER_LOG_VERSION 1 +#define CONFIG_BOOTLOADER_LOG_LEVEL_INFO 1 +#define CONFIG_BOOTLOADER_LOG_LEVEL 3 +#define CONFIG_BOOTLOADER_LOG_TIMESTAMP_SOURCE_CPU_TICKS 1 +#define CONFIG_BOOTLOADER_LOG_MODE_TEXT_EN 1 +#define CONFIG_BOOTLOADER_LOG_MODE_TEXT 1 +#define CONFIG_BOOTLOADER_CPU_CLK_FREQ_MHZ 90 +#define CONFIG_BOOTLOADER_FLASH_XMC_SUPPORT 1 +#define CONFIG_BOOTLOADER_REGION_PROTECTION_ENABLE 1 +#define CONFIG_BOOTLOADER_WDT_ENABLE 1 +#define CONFIG_BOOTLOADER_WDT_TIME_MS 9000 +#define CONFIG_BOOTLOADER_RESERVE_RTC_SIZE 0x0 +#define CONFIG_SECURE_BOOT_V2_RSA_SUPPORTED 1 +#define CONFIG_SECURE_BOOT_V2_ECC_SUPPORTED 1 +#define CONFIG_SECURE_BOOT_V2_ECDSA_INSECURE 1 +#define CONFIG_SECURE_BOOT_V2_PREFERRED 1 +#define CONFIG_SECURE_ROM_DL_MODE_ENABLED 1 +#define CONFIG_APP_COMPILE_TIME_DATE 1 +#define CONFIG_APP_RETRIEVE_LEN_ELF_SHA 9 +#define CONFIG_ESP_ROM_HAS_CRC_LE 1 +#define CONFIG_ESP_ROM_HAS_CRC_BE 1 +#define CONFIG_ESP_ROM_UART_CLK_IS_XTAL 1 +#define CONFIG_ESP_ROM_USB_SERIAL_DEVICE_NUM 6 +#define CONFIG_ESP_ROM_USB_OTG_NUM 5 +#define CONFIG_ESP_ROM_HAS_RETARGETABLE_LOCKING 1 +#define CONFIG_ESP_ROM_GET_CLK_FREQ 1 +#define CONFIG_ESP_ROM_HAS_RVFPLIB 1 +#define CONFIG_ESP_ROM_HAS_HAL_WDT 1 +#define CONFIG_ESP_ROM_HAS_HAL_SYSTIMER 1 +#define CONFIG_ESP_ROM_SYSTIMER_INIT_PATCH 1 +#define CONFIG_ESP_ROM_HAS_LAYOUT_TABLE 1 +#define CONFIG_ESP_ROM_WDT_INIT_PATCH 1 +#define CONFIG_ESP_ROM_HAS_LP_ROM 1 +#define CONFIG_ESP_ROM_WITHOUT_REGI2C 1 +#define CONFIG_ESP_ROM_HAS_NEWLIB 1 +#define CONFIG_ESP_ROM_HAS_NEWLIB_NANO_FORMAT 1 +#define CONFIG_ESP_ROM_HAS_NEWLIB_NANO_PRINTF_FLOAT_BUG 1 +#define CONFIG_ESP_ROM_HAS_VERSION 1 +#define CONFIG_ESP_ROM_CLIC_INT_TYPE_PATCH 1 +#define CONFIG_ESP_ROM_HAS_OUTPUT_PUTC_FUNC 1 +#define CONFIG_ESP_ROM_HAS_SUBOPTIMAL_NEWLIB_ON_MISALIGNED_MEMORY 1 +#define CONFIG_ESP_ROM_ECDSA_VERIFY_PATCH 1 +#define CONFIG_ESP_ROM_BOOTLOADER_OFFSET_FLASH 0x2000 +#define CONFIG_ESP_ROM_CACHE_WRITEBACK_NEEDS_SYNC_TWICE_MAP 1 +#define CONFIG_BOOT_ROM_LOG_ALWAYS_ON 1 +#define CONFIG_ESPTOOLPY_FLASHMODE_DIO 1 +#define CONFIG_ESPTOOLPY_FLASH_SAMPLE_MODE_STR 1 +#define CONFIG_ESPTOOLPY_FLASHMODE "dio" +#define CONFIG_ESPTOOLPY_FLASHFREQ_80M 1 +#define CONFIG_ESPTOOLPY_FLASHFREQ_VAL 80 +#define CONFIG_ESPTOOLPY_FLASHFREQ "80m" +#define CONFIG_ESPTOOLPY_FLASHSIZE_16MB 1 +#define CONFIG_ESPTOOLPY_FLASHSIZE "16MB" +#define CONFIG_ESPTOOLPY_BEFORE_RESET 1 +#define CONFIG_ESPTOOLPY_BEFORE "default-reset" +#define CONFIG_ESPTOOLPY_AFTER_RESET 1 +#define CONFIG_ESPTOOLPY_AFTER "hard-reset" +#define CONFIG_ESPTOOLPY_MONITOR_BAUD 115200 +#define CONFIG_PARTITION_TABLE_SINGLE_APP_LARGE 1 +#define CONFIG_PARTITION_TABLE_CUSTOM_FILENAME "partitions.csv" +#define CONFIG_PARTITION_TABLE_FILENAME "partitions_singleapp_large.csv" +#define CONFIG_PARTITION_TABLE_OFFSET 0x8000 +#define CONFIG_PARTITION_TABLE_MD5 1 +#define CONFIG_COMPILER_OPTIMIZATION_DEBUG 1 +#define CONFIG_COMPILER_OPTIMIZATION_ASSERTIONS_ENABLE 1 +#define CONFIG_COMPILER_FLOAT_LIB_FROM_RVFPLIB 1 +#define CONFIG_COMPILER_OPTIMIZATION_ASSERTION_LEVEL 2 +#define CONFIG_COMPILER_HIDE_PATHS_MACROS 1 +#define CONFIG_COMPILER_STACK_CHECK_MODE_NONE 1 +#define CONFIG_COMPILER_RT_LIB_GCCLIB 1 +#define CONFIG_COMPILER_RT_LIB_NAME "gcc" +#define CONFIG_COMPILER_ORPHAN_SECTIONS_ERROR 1 +#define CONFIG_COMPILER_CXX_GLIBCXX_CONSTEXPR_NO_CHANGE 1 +#define CONFIG_BT_ENABLED 1 +#define CONFIG_BT_NIMBLE_ENABLED 1 +#define CONFIG_BT_CONTROLLER_DISABLED 1 +#define CONFIG_BT_ALARM_MAX_NUM 50 +#define CONFIG_BT_SMP_CRYPTO_STACK_TINYCRYPT 1 +#define CONFIG_BT_NIMBLE_MEM_ALLOC_MODE_INTERNAL 1 +#define CONFIG_BT_NIMBLE_PINNED_TO_CORE 0 +#define CONFIG_BT_NIMBLE_PINNED_TO_CORE_0 1 +#define CONFIG_BT_NIMBLE_HOST_TASK_STACK_SIZE 4096 +#define CONFIG_BT_NIMBLE_ROLE_PERIPHERAL 1 +#define CONFIG_BT_NIMBLE_ROLE_BROADCASTER 1 +#define CONFIG_BT_NIMBLE_ROLE_OBSERVER 1 +#define CONFIG_BT_NIMBLE_GATT_SERVER 1 +#define CONFIG_BT_NIMBLE_SECURITY_ENABLE 1 +#define CONFIG_BT_NIMBLE_SM_LEGACY 1 +#define CONFIG_BT_NIMBLE_SM_SC 1 +#define CONFIG_BT_NIMBLE_LL_CFG_FEAT_LE_ENCRYPTION 1 +#define CONFIG_BT_NIMBLE_SM_LVL 0 +#define CONFIG_BT_NIMBLE_SM_SC_ONLY 0 +#define CONFIG_BT_NIMBLE_MAX_BONDS 3 +#define CONFIG_BT_NIMBLE_RPA_TIMEOUT 900 +#define CONFIG_BT_NIMBLE_WHITELIST_SIZE 12 +#define CONFIG_BT_NIMBLE_HS_PVCY 1 +#define CONFIG_BT_NIMBLE_MAX_CONNECTIONS 3 +#define CONFIG_BT_NIMBLE_MAX_CCCDS 8 +#define CONFIG_BT_NIMBLE_HS_STOP_TIMEOUT_MS 2000 +#define CONFIG_BT_NIMBLE_USE_ESP_TIMER 1 +#define CONFIG_BT_NIMBLE_ATT_PREFERRED_MTU 256 +#define CONFIG_BT_NIMBLE_ATT_MAX_PREP_ENTRIES 64 +#define CONFIG_BT_NIMBLE_GATT_MAX_PROCS 4 +#define CONFIG_BT_NIMBLE_L2CAP_COC_MAX_NUM 0 +#define CONFIG_BT_NIMBLE_MSYS_1_BLOCK_COUNT 12 +#define CONFIG_BT_NIMBLE_MSYS_1_BLOCK_SIZE 256 +#define CONFIG_BT_NIMBLE_MSYS_2_BLOCK_COUNT 24 +#define CONFIG_BT_NIMBLE_MSYS_2_BLOCK_SIZE 320 +#define CONFIG_BT_NIMBLE_TRANSPORT_ACL_FROM_LL_COUNT 24 +#define CONFIG_BT_NIMBLE_TRANSPORT_ACL_SIZE 255 +#define CONFIG_BT_NIMBLE_TRANSPORT_EVT_SIZE 70 +#define CONFIG_BT_NIMBLE_TRANSPORT_EVT_COUNT 30 +#define CONFIG_BT_NIMBLE_TRANSPORT_EVT_DISCARD_COUNT 8 +#define CONFIG_BT_NIMBLE_L2CAP_COC_SDU_BUFF_COUNT 1 +#define CONFIG_BT_NIMBLE_PROX_SERVICE 1 +#define CONFIG_BT_NIMBLE_ANS_SERVICE 1 +#define CONFIG_BT_NIMBLE_CTS_SERVICE 1 +#define CONFIG_BT_NIMBLE_HTP_SERVICE 1 +#define CONFIG_BT_NIMBLE_IPSS_SERVICE 1 +#define CONFIG_BT_NIMBLE_TPS_SERVICE 1 +#define CONFIG_BT_NIMBLE_IAS_SERVICE 1 +#define CONFIG_BT_NIMBLE_LLS_SERVICE 1 +#define CONFIG_BT_NIMBLE_SPS_SERVICE 1 +#define CONFIG_BT_NIMBLE_HR_SERVICE 1 +#define CONFIG_BT_NIMBLE_BAS_SERVICE 1 +#define CONFIG_BT_NIMBLE_DIS_SERVICE 1 +#define CONFIG_BT_NIMBLE_GAP_SERVICE 1 +#define CONFIG_BT_NIMBLE_SVC_GAP_DEVICE_NAME "nimble" +#define CONFIG_BT_NIMBLE_GAP_DEVICE_NAME_MAX_LEN 31 +#define CONFIG_BT_NIMBLE_SVC_GAP_APPEARANCE 0x0 +#define CONFIG_BT_NIMBLE_SVC_GAP_NAME_WRITE_PERM 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_NAME_WRITE_PERM_ENC 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_NAME_WRITE_PERM_AUTHEN 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_NAME_WRITE_PERM_AUTHOR 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_CAR_CHAR_NOT_SUPP 1 +#define CONFIG_BT_NIMBLE_SVC_GAP_CENT_ADDR_RESOLUTION -1 +#define CONFIG_BT_NIMBLE_SVC_GAP_APPEAR_WRITE_PERM 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_APPEAR_WRITE_PERM_ENC 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_APPEAR_WRITE_PERM_ATHN 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_APPEAR_WRITE_PERM_ATHR 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_PPCP_MAX_CONN_INTERVAL 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_PPCP_MIN_CONN_INTERVAL 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_PPCP_SLAVE_LATENCY 0 +#define CONFIG_BT_NIMBLE_SVC_GAP_PPCP_SUPERVISION_TMO 0 +#define CONFIG_BT_NIMBLE_EATT_CHAN_NUM 0 +#define CONFIG_BT_NIMBLE_DTM_MODE_TEST 1 +#define CONFIG_BT_NIMBLE_MEM_OPTIMIZATION 1 +#define CONFIG_BT_NIMBLE_STATIC_TO_DYNAMIC 1 +#define CONFIG_BT_NIMBLE_SM_SIGN_CNT 1 +#define CONFIG_BT_NIMBLE_CPFD_CAFD 1 +#define CONFIG_BT_NIMBLE_RECONFIG_MTU 1 +#define CONFIG_UART_HW_FLOWCTRL_DISABLE 1 +#define CONFIG_BT_NIMBLE_HCI_UART_FLOW_CTRL 0 +#define CONFIG_BT_NIMBLE_HCI_UART_RTS_PIN 19 +#define CONFIG_BT_NIMBLE_HCI_UART_CTS_PIN 23 +#define CONFIG_BT_NIMBLE_LOG_LEVEL_INFO 1 +#define CONFIG_BT_NIMBLE_LOG_LEVEL 1 +#define CONFIG_BT_NIMBLE_PRINT_ERR_NAME 1 +#define CONFIG_BT_NIMBLE_CHK_HOST_STATUS 1 +#define CONFIG_BT_NIMBLE_UTIL_API 1 +#define CONFIG_BT_NIMBLE_EXTRA_ADV_FIELDS 1 +#define CONFIG_EFUSE_MAX_BLK_LEN 256 +#define CONFIG_ESP_TLS_USING_MBEDTLS 1 +#define CONFIG_ESP_TLS_USE_DS_PERIPHERAL 1 +#define CONFIG_ESP_TLS_DYN_BUF_STRATEGY_SUPPORTED 1 +#define CONFIG_ESP_ERR_TO_NAME_LOOKUP 1 +#define CONFIG_ANA_CMPR_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_ANA_CMPR_OBJ_CACHE_SAFE 1 +#define CONFIG_GDMA_CTRL_FUNC_IN_IRAM 1 +#define CONFIG_GDMA_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_GDMA_OBJ_DRAM_SAFE 1 +#define CONFIG_GPTIMER_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_GPTIMER_OBJ_CACHE_SAFE 1 +#define CONFIG_I2C_MASTER_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_MCPWM_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_MCPWM_OBJ_CACHE_SAFE 1 +#define CONFIG_PARLIO_TX_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_PARLIO_RX_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_PARLIO_OBJ_CACHE_SAFE 1 +#define CONFIG_RMT_ENCODER_FUNC_IN_IRAM 1 +#define CONFIG_RMT_TX_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_RMT_RX_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_RMT_OBJ_CACHE_SAFE 1 +#define CONFIG_SPI_MASTER_ISR_IN_IRAM 1 +#define CONFIG_SPI_SLAVE_ISR_IN_IRAM 1 +#define CONFIG_USJ_ENABLE_USB_SERIAL_JTAG 1 +#define CONFIG_ETH_ENABLED 1 +#define CONFIG_ETH_USE_ESP32_EMAC 1 +#define CONFIG_ETH_DMA_BUFFER_SIZE 512 +#define CONFIG_ETH_DMA_RX_BUFFER_NUM 20 +#define CONFIG_ETH_DMA_TX_BUFFER_NUM 10 +#define CONFIG_ETH_USE_SPI_ETHERNET 1 +#define CONFIG_ESP_EVENT_POST_FROM_ISR 1 +#define CONFIG_ESP_EVENT_POST_FROM_IRAM_ISR 1 +#define CONFIG_ESP_GDBSTUB_ENABLED 1 +#define CONFIG_ESP_GDBSTUB_SUPPORT_TASKS 1 +#define CONFIG_ESP_GDBSTUB_MAX_TASKS 32 +#define CONFIG_ESPHID_TASK_SIZE_BT 2048 +#define CONFIG_ESPHID_TASK_SIZE_BLE 4096 +#define CONFIG_ESP_HTTP_CLIENT_ENABLE_HTTPS 1 +#define CONFIG_ESP_HTTP_CLIENT_EVENT_POST_TIMEOUT 2000 +#define CONFIG_HTTPD_MAX_REQ_HDR_LEN 1024 +#define CONFIG_HTTPD_MAX_URI_LEN 512 +#define CONFIG_HTTPD_ERR_RESP_NO_DELAY 1 +#define CONFIG_HTTPD_PURGE_BUF_LEN 32 +#define CONFIG_HTTPD_SERVER_EVENT_POST_TIMEOUT 2000 +#define CONFIG_ESP_HTTPS_OTA_EVENT_POST_TIMEOUT 2000 +#define CONFIG_ESP_HTTPS_SERVER_EVENT_POST_TIMEOUT 2000 +#define CONFIG_ESP_HW_SUPPORT_FUNC_IN_IRAM 1 +#define CONFIG_ESP32P4_SELECTS_REV_LESS_V3 1 +#define CONFIG_ESP32P4_REV_MIN_100 1 +#define CONFIG_ESP32P4_REV_MIN_FULL 100 +#define CONFIG_ESP_REV_MIN_FULL 100 +#define CONFIG_ESP32P4_REV_MAX_FULL 199 +#define CONFIG_ESP_REV_MAX_FULL 199 +#define CONFIG_ESP_EFUSE_BLOCK_REV_MIN_FULL 0 +#define CONFIG_ESP_EFUSE_BLOCK_REV_MAX_FULL 199 +#define CONFIG_ESP_MAC_ADDR_UNIVERSE_ETH 1 +#define CONFIG_ESP_MAC_UNIVERSAL_MAC_ADDRESSES_ONE 1 +#define CONFIG_ESP_MAC_UNIVERSAL_MAC_ADDRESSES 1 +#define CONFIG_ESP32P4_UNIVERSAL_MAC_ADDRESSES_ONE 1 +#define CONFIG_ESP32P4_UNIVERSAL_MAC_ADDRESSES 1 +#define CONFIG_ESP_SLEEP_FLASH_LEAKAGE_WORKAROUND 1 +#define CONFIG_ESP_SLEEP_PSRAM_LEAKAGE_WORKAROUND 1 +#define CONFIG_ESP_SLEEP_GPIO_RESET_WORKAROUND 1 +#define CONFIG_ESP_SLEEP_WAIT_FLASH_READY_EXTRA_DELAY 0 +#define CONFIG_ESP_SLEEP_GPIO_ENABLE_INTERNAL_RESISTORS 1 +#define CONFIG_RTC_CLK_SRC_INT_RC 1 +#define CONFIG_RTC_CLK_CAL_CYCLES 1024 +#define CONFIG_RTC_FAST_CLK_SRC_RC_FAST 1 +#define CONFIG_RTC_CLK_FUNC_IN_IRAM 1 +#define CONFIG_RTC_TIME_FUNC_IN_IRAM 1 +#define CONFIG_ESP_PERIPH_CTRL_FUNC_IN_IRAM 1 +#define CONFIG_ESP_REGI2C_CTRL_FUNC_IN_IRAM 1 +#define CONFIG_XTAL_FREQ_40 1 +#define CONFIG_XTAL_FREQ 40 +#define CONFIG_ESP_SLEEP_DCM_VSET_VAL_IN_SLEEP 14 +#define CONFIG_ESP_LDO_RESERVE_SPI_NOR_FLASH 1 +#define CONFIG_ESP_LDO_CHAN_SPI_NOR_FLASH_DOMAIN 1 +#define CONFIG_ESP_LDO_VOLTAGE_SPI_NOR_FLASH_3300_MV 1 +#define CONFIG_ESP_LDO_VOLTAGE_SPI_NOR_FLASH_DOMAIN 3300 +#define CONFIG_ESP_LDO_RESERVE_PSRAM 1 +#define CONFIG_ESP_LDO_CHAN_PSRAM_DOMAIN 2 +#define CONFIG_ESP_LDO_VOLTAGE_PSRAM_1800_MV 1 +#define CONFIG_ESP_LDO_VOLTAGE_PSRAM_DOMAIN 1800 +#define CONFIG_ESP_BROWNOUT_DET 1 +#define CONFIG_ESP_BROWNOUT_DET_LVL_SEL_7 1 +#define CONFIG_ESP_BROWNOUT_DET_LVL 7 +#define CONFIG_ESP_BROWNOUT_USE_INTR 1 +#define CONFIG_ESP_SPI_BUS_LOCK_ISR_FUNCS_IN_IRAM 1 +#define CONFIG_ESP_ENABLE_PVT 1 +#define CONFIG_ESP_INTR_IN_IRAM 1 +#define CONFIG_P4_REV3_MSPI_WORKAROUND_SIZE 0x0 +#define CONFIG_LCD_DSI_ISR_HANDLER_IN_IRAM 1 +#define CONFIG_LCD_DSI_OBJ_FORCE_INTERNAL 1 +#define CONFIG_LIBC_PICOLIBC 1 +#define CONFIG_LIBC_PICOLIBC_NEWLIB_COMPATIBILITY 1 +#define CONFIG_LIBC_MISC_IN_IRAM 1 +#define CONFIG_LIBC_LOCKS_PLACE_IN_IRAM 1 +#define CONFIG_LIBC_STDOUT_LINE_ENDING_CRLF 1 +#define CONFIG_LIBC_STDIN_LINE_ENDING_CR 1 +#define CONFIG_LIBC_TIME_SYSCALL_USE_RTC_HRT 1 +#define CONFIG_LIBC_OPTIMIZED_MISALIGNED_ACCESS 1 +#define CONFIG_LIBC_ASSERT_BUFFER_SIZE 200 +#define CONFIG_ESP_NETIF_LOST_IP_TIMER_ENABLE 1 +#define CONFIG_ESP_NETIF_IP_LOST_TIMER_INTERVAL 120 +#define CONFIG_ESP_NETIF_TCPIP_LWIP 1 +#define CONFIG_ESP_NETIF_USES_TCPIP_WITH_BSD_API 1 +#define CONFIG_ESP_NETIF_REPORT_DATA_TRAFFIC 1 +#define CONFIG_ESP_NETIF_RECEIVE_REPORT_ERRORS 1 +#define CONFIG_PM_SLEEP_FUNC_IN_IRAM 1 +#define CONFIG_PM_SLP_IRAM_OPT 1 +#define CONFIG_PM_SLP_DEFAULT_PARAMS_OPT 1 +#define CONFIG_SPIRAM 1 +#define CONFIG_SPIRAM_MODE_HEX 1 +#define CONFIG_SPIRAM_SPEED_200M 1 +#define CONFIG_SPIRAM_SPEED 200 +#define CONFIG_SPIRAM_BOOT_HW_INIT 1 +#define CONFIG_SPIRAM_BOOT_INIT 1 +#define CONFIG_SPIRAM_PRE_CONFIGURE_MEMORY_PROTECTION 1 +#define CONFIG_SPIRAM_USE_MALLOC 1 +#define CONFIG_SPIRAM_MEMTEST 1 +#define CONFIG_SPIRAM_MALLOC_ALWAYSINTERNAL 16384 +#define CONFIG_SPIRAM_MALLOC_RESERVE_INTERNAL 32768 +#define CONFIG_ESP_ROM_PRINT_IN_IRAM 1 +#define CONFIG_ESP_CONSOLE_UART_DEFAULT 1 +#define CONFIG_ESP_CONSOLE_SECONDARY_USB_SERIAL_JTAG 1 +#define CONFIG_ESP_CONSOLE_USB_SERIAL_JTAG_ENABLED 1 +#define CONFIG_ESP_CONSOLE_UART 1 +#define CONFIG_ESP_CONSOLE_UART_NUM 0 +#define CONFIG_ESP_CONSOLE_ROM_SERIAL_PORT_NUM 0 +#define CONFIG_ESP_CONSOLE_UART_BAUDRATE 115200 +#define CONFIG_ESP_DEFAULT_CPU_FREQ_MHZ_360 1 +#define CONFIG_ESP_DEFAULT_CPU_FREQ_MHZ 360 +#define CONFIG_CACHE_L2_CACHE_128KB 1 +#define CONFIG_CACHE_L2_CACHE_SIZE 0x20000 +#define CONFIG_CACHE_L2_CACHE_LINE_64B 1 +#define CONFIG_CACHE_L2_CACHE_LINE_SIZE 64 +#define CONFIG_CACHE_L1_CACHE_LINE_SIZE 64 +#define CONFIG_ESP_SYSTEM_IN_IRAM 1 +#define CONFIG_ESP_SYSTEM_PANIC_PRINT_REBOOT 1 +#define CONFIG_ESP_SYSTEM_PANIC_REBOOT_DELAY_SECONDS 0 +#define CONFIG_ESP_SYSTEM_RTC_FAST_MEM_AS_HEAP_DEPCHECK 1 +#define CONFIG_ESP_SYSTEM_ALLOW_RTC_FAST_MEM_AS_HEAP 1 +#define CONFIG_ESP_SYSTEM_NO_BACKTRACE 1 +#define CONFIG_ESP_SYSTEM_MEMPROT 1 +#define CONFIG_ESP_SYSTEM_MEMPROT_PMP 1 +#define CONFIG_ESP_SYSTEM_EVENT_QUEUE_SIZE 32 +#define CONFIG_ESP_SYSTEM_EVENT_TASK_STACK_SIZE 2304 +#define CONFIG_ESP_MAIN_TASK_STACK_SIZE 6144 +#define CONFIG_ESP_MAIN_TASK_AFFINITY_CPU0 1 +#define CONFIG_ESP_MAIN_TASK_AFFINITY 0x0 +#define CONFIG_ESP_MINIMAL_SHARED_STACK_SIZE 2048 +#define CONFIG_ESP_INT_WDT 1 +#define CONFIG_ESP_INT_WDT_TIMEOUT_MS 300 +#define CONFIG_ESP_INT_WDT_CHECK_CPU1 1 +#define CONFIG_ESP_TASK_WDT_EN 1 +#define CONFIG_ESP_TASK_WDT_INIT 1 +#define CONFIG_ESP_TASK_WDT_TIMEOUT_S 5 +#define CONFIG_ESP_TASK_WDT_CHECK_IDLE_TASK_CPU0 1 +#define CONFIG_ESP_TASK_WDT_CHECK_IDLE_TASK_CPU1 1 +#define CONFIG_ESP_DEBUG_OCDAWARE 1 +#define CONFIG_ESP_SYSTEM_CHECK_INT_LEVEL_4 1 +#define CONFIG_ESP_SYSTEM_HW_STACK_GUARD 1 +#define CONFIG_ESP_SYSTEM_HW_PC_RECORD 1 +#define CONFIG_ESP_IPC_ENABLE 1 +#define CONFIG_ESP_IPC_TASK_STACK_SIZE 1024 +#define CONFIG_ESP_IPC_USES_CALLERS_PRIORITY 1 +#define CONFIG_ESP_IPC_ISR_ENABLE 1 +#define CONFIG_ESP_TIMER_IN_IRAM 1 +#define CONFIG_ESP_TIME_FUNCS_USE_RTC_TIMER 1 +#define CONFIG_ESP_TIME_FUNCS_USE_ESP_TIMER 1 +#define CONFIG_ESP_TIMER_TASK_STACK_SIZE 3584 +#define CONFIG_ESP_TIMER_INTERRUPT_LEVEL 1 +#define CONFIG_ESP_TIMER_TASK_AFFINITY 0x0 +#define CONFIG_ESP_TIMER_TASK_AFFINITY_CPU0 1 +#define CONFIG_ESP_TIMER_ISR_AFFINITY_CPU0 1 +#define CONFIG_ESP_TIMER_IMPL_SYSTIMER 1 +#define CONFIG_ESP_TRACE_LIB_NONE 1 +#define CONFIG_ESP_TRACE_LIB_NAME "none" +#define CONFIG_ESP_TRACE_TRANSPORT_NONE 1 +#define CONFIG_ESP_TRACE_TRANSPORT_NAME "none" +#define CONFIG_ESP_WIFI_STATIC_RX_BUFFER_NUM 10 +#define CONFIG_ESP_WIFI_DYNAMIC_RX_BUFFER_NUM 32 +#define CONFIG_ESP_WIFI_TX_BUFFER_TYPE 1 +#define CONFIG_ESP_WIFI_DYNAMIC_TX_BUFFER_NUM 32 +#define CONFIG_ESP_WIFI_DYNAMIC_RX_MGMT_BUF 0 +#define CONFIG_ESP_WIFI_RX_MGMT_BUF_NUM_DEF 5 +#define CONFIG_ESP_WIFI_AMPDU_TX_ENABLED 1 +#define CONFIG_ESP_WIFI_TX_BA_WIN 6 +#define CONFIG_ESP_WIFI_AMPDU_RX_ENABLED 1 +#define CONFIG_ESP_WIFI_RX_BA_WIN 6 +#define CONFIG_ESP_WIFI_NVS_ENABLED 1 +#define CONFIG_ESP_WIFI_SOFTAP_BEACON_MAX_LEN 752 +#define CONFIG_ESP_WIFI_MGMT_SBUF_NUM 32 +#define CONFIG_ESP_WIFI_IRAM_OPT 1 +#define CONFIG_ESP_WIFI_EXTRA_IRAM_OPT 1 +#define CONFIG_ESP_WIFI_RX_IRAM_OPT 1 +#define CONFIG_ESP_WIFI_ENABLE_WPA3_SAE 1 +#define CONFIG_ESP_WIFI_ENABLE_SAE_H2E 1 +#define CONFIG_ESP_WIFI_ENABLE_SAE_PK 1 +#define CONFIG_ESP_WIFI_SOFTAP_SAE_SUPPORT 1 +#define CONFIG_ESP_WIFI_ENABLE_WPA3_OWE_STA 1 +#define CONFIG_ESP_WIFI_WPA3_COMPATIBLE_SUPPORT 1 +#define CONFIG_ESP_WIFI_SLP_IRAM_OPT 1 +#define CONFIG_ESP_WIFI_SLP_DEFAULT_MIN_ACTIVE_TIME 50 +#define CONFIG_ESP_WIFI_BSS_MAX_IDLE_SUPPORT 1 +#define CONFIG_ESP_WIFI_SLP_DEFAULT_MAX_ACTIVE_TIME 10 +#define CONFIG_ESP_WIFI_SLP_DEFAULT_WAIT_BROADCAST_DATA_TIME 15 +#define CONFIG_ESP_WIFI_STA_DISCONNECTED_PM_ENABLE 1 +#define CONFIG_ESP_WIFI_GMAC_SUPPORT 1 +#define CONFIG_ESP_WIFI_SOFTAP_SUPPORT 1 +#define CONFIG_ESP_WIFI_ESPNOW_MAX_ENCRYPT_NUM 7 +#define CONFIG_ESP_WIFI_MBEDTLS_CRYPTO 1 +#define CONFIG_ESP_WIFI_MBEDTLS_TLS_CLIENT 1 +#define CONFIG_ESP_WIFI_TX_HETB_QUEUE_NUM 3 +#define CONFIG_ESP_WIFI_ENTERPRISE_SUPPORT 1 +#define CONFIG_ESP_COREDUMP_ENABLE_TO_NONE 1 +#define CONFIG_FATFS_VOLUME_COUNT 2 +#define CONFIG_FATFS_LFN_HEAP 1 +#define CONFIG_FATFS_SECTOR_4096 1 +#define CONFIG_FATFS_CODEPAGE_437 1 +#define CONFIG_FATFS_CODEPAGE 437 +#define CONFIG_FATFS_MAX_LFN 255 +#define CONFIG_FATFS_API_ENCODING_ANSI_OEM 1 +#define CONFIG_FATFS_FS_LOCK 0 +#define CONFIG_FATFS_TIMEOUT_MS 10000 +#define CONFIG_FATFS_PER_FILE_CACHE 1 +#define CONFIG_FATFS_ALLOC_PREFER_EXTRAM 1 +#define CONFIG_FATFS_USE_STRFUNC_NONE 1 +#define CONFIG_FATFS_VFS_FSTAT_BLKSIZE 0 +#define CONFIG_FATFS_LINK_LOCK 1 +#define CONFIG_FATFS_USE_DYN_BUFFERS 1 +#define CONFIG_FATFS_DONT_TRUST_FREE_CLUSTER_CNT 0 +#define CONFIG_FATFS_DONT_TRUST_LAST_ALLOC 0 +#define CONFIG_FREERTOS_HZ 1000 +#define CONFIG_FREERTOS_CHECK_STACKOVERFLOW_CANARY 1 +#define CONFIG_FREERTOS_THREAD_LOCAL_STORAGE_POINTERS 1 +#define CONFIG_FREERTOS_IDLE_TASK_STACKSIZE 1536 +#define CONFIG_FREERTOS_MAX_TASK_NAME_LEN 16 +#define CONFIG_FREERTOS_USE_TIMERS 1 +#define CONFIG_FREERTOS_TIMER_SERVICE_TASK_NAME "Tmr Svc" +#define CONFIG_FREERTOS_TIMER_TASK_NO_AFFINITY 1 +#define CONFIG_FREERTOS_TIMER_SERVICE_TASK_CORE_AFFINITY 0x7FFFFFFF +#define CONFIG_FREERTOS_TIMER_TASK_PRIORITY 1 +#define CONFIG_FREERTOS_TIMER_TASK_STACK_DEPTH 2048 +#define CONFIG_FREERTOS_TIMER_QUEUE_LENGTH 10 +#define CONFIG_FREERTOS_QUEUE_REGISTRY_SIZE 0 +#define CONFIG_FREERTOS_TASK_NOTIFICATION_ARRAY_ENTRIES 1 +#define CONFIG_FREERTOS_TASK_FUNCTION_WRAPPER 1 +#define CONFIG_FREERTOS_TLSP_DELETION_CALLBACKS 1 +#define CONFIG_FREERTOS_CHECK_MUTEX_GIVEN_BY_OWNER 1 +#define CONFIG_FREERTOS_ISR_STACKSIZE 1536 +#define CONFIG_FREERTOS_INTERRUPT_BACKTRACE 1 +#define CONFIG_FREERTOS_TICK_SUPPORT_SYSTIMER 1 +#define CONFIG_FREERTOS_CORETIMER_SYSTIMER_LVL1 1 +#define CONFIG_FREERTOS_SYSTICK_USES_SYSTIMER 1 +#define CONFIG_FREERTOS_TASK_CREATE_ALLOW_EXT_MEM 1 +#define CONFIG_FREERTOS_PORT 1 +#define CONFIG_FREERTOS_NO_AFFINITY 0x7FFFFFFF +#define CONFIG_FREERTOS_SUPPORT_STATIC_ALLOCATION 1 +#define CONFIG_FREERTOS_DEBUG_OCDAWARE 1 +#define CONFIG_FREERTOS_NUMBER_OF_CORES 2 +#define CONFIG_HAL_ASSERTION_EQUALS_SYSTEM 1 +#define CONFIG_HAL_DEFAULT_ASSERTION_LEVEL 2 +#define CONFIG_HAL_SYSTIMER_USE_ROM_IMPL 1 +#define CONFIG_HAL_WDT_USE_ROM_IMPL 1 +#define CONFIG_HAL_GPIO_USE_ROM_IMPL 1 +#define CONFIG_HEAP_POISONING_DISABLED 1 +#define CONFIG_HEAP_TRACING_OFF 1 +#define CONFIG_LOG_VERSION_1 1 +#define CONFIG_LOG_VERSION 1 +#define CONFIG_LOG_DEFAULT_LEVEL_INFO 1 +#define CONFIG_LOG_DEFAULT_LEVEL 3 +#define CONFIG_LOG_MAXIMUM_EQUALS_DEFAULT 1 +#define CONFIG_LOG_MAXIMUM_LEVEL 3 +#define CONFIG_LOG_DYNAMIC_LEVEL_CONTROL 1 +#define CONFIG_LOG_TAG_LEVEL_IMPL_CACHE_AND_LINKED_LIST 1 +#define CONFIG_LOG_TAG_LEVEL_CACHE_BINARY_MIN_HEAP 1 +#define CONFIG_LOG_TAG_LEVEL_IMPL_CACHE_SIZE 31 +#define CONFIG_LOG_TIMESTAMP_SOURCE_RTOS 1 +#define CONFIG_LOG_MODE_TEXT_EN 1 +#define CONFIG_LOG_MODE_TEXT 1 +#define CONFIG_LOG_IN_IRAM 1 +#define CONFIG_LWIP_ENABLE 1 +#define CONFIG_LWIP_LOCAL_HOSTNAME "espressif" +#define CONFIG_LWIP_TCPIP_TASK_PRIO 18 +#define CONFIG_LWIP_DNS_SUPPORT_MDNS_QUERIES 1 +#define CONFIG_LWIP_TIMERS_ONDEMAND 1 +#define CONFIG_LWIP_ND6 1 +#define CONFIG_LWIP_MAX_SOCKETS 10 +#define CONFIG_LWIP_SO_REUSE 1 +#define CONFIG_LWIP_SO_REUSE_RXTOALL 1 +#define CONFIG_LWIP_IP_DEFAULT_TTL 64 +#define CONFIG_LWIP_IP4_FRAG 1 +#define CONFIG_LWIP_IP6_FRAG 1 +#define CONFIG_LWIP_IP_REASS_MAX_PBUFS 10 +#define CONFIG_LWIP_IPV6_DUP_DETECT_ATTEMPTS 1 +#define CONFIG_LWIP_ESP_GRATUITOUS_ARP 1 +#define CONFIG_LWIP_GARP_TMR_INTERVAL 60 +#define CONFIG_LWIP_ESP_MLDV6_REPORT 1 +#define CONFIG_LWIP_MLDV6_TMR_INTERVAL 40 +#define CONFIG_LWIP_TCPIP_RECVMBOX_SIZE 32 +#define CONFIG_LWIP_DHCP_DOES_ARP_CHECK 1 +#define CONFIG_LWIP_DHCP_DISABLE_VENDOR_CLASS_ID 1 +#define CONFIG_LWIP_DHCP_OPTIONS_LEN 69 +#define CONFIG_LWIP_NUM_NETIF_CLIENT_DATA 0 +#define CONFIG_LWIP_DHCP_COARSE_TIMER_SECS 1 +#define CONFIG_LWIP_DHCPS 1 +#define CONFIG_LWIP_DHCPS_REPORT_CLIENT_HOSTNAME 1 +#define CONFIG_LWIP_DHCPS_LEASE_UNIT 60 +#define CONFIG_LWIP_DHCPS_MAX_STATION_NUM 8 +#define CONFIG_LWIP_DHCPS_MAX_HOSTNAME_LEN 64 +#define CONFIG_LWIP_DHCPS_STATIC_ENTRIES 1 +#define CONFIG_LWIP_IPV4 1 +#define CONFIG_LWIP_IPV6 1 +#define CONFIG_LWIP_IPV6_NUM_ADDRESSES 3 +#define CONFIG_LWIP_NETIF_LOOPBACK 1 +#define CONFIG_LWIP_LOOPBACK_MAX_PBUFS 8 +#define CONFIG_LWIP_MAX_ACTIVE_TCP 16 +#define CONFIG_LWIP_MAX_LISTENING_TCP 16 +#define CONFIG_LWIP_TCP_HIGH_SPEED_RETRANSMISSION 1 +#define CONFIG_LWIP_TCP_MAXRTX 12 +#define CONFIG_LWIP_TCP_SYNMAXRTX 12 +#define CONFIG_LWIP_TCP_MSS 1440 +#define CONFIG_LWIP_TCP_TMR_INTERVAL 250 +#define CONFIG_LWIP_TCP_MSL 60000 +#define CONFIG_LWIP_TCP_FIN_WAIT_TIMEOUT 20000 +#define CONFIG_LWIP_TCP_SND_BUF_DEFAULT 5760 +#define CONFIG_LWIP_TCP_WND_DEFAULT 5760 +#define CONFIG_LWIP_TCP_RECVMBOX_SIZE 6 +#define CONFIG_LWIP_TCP_ACCEPTMBOX_SIZE 6 +#define CONFIG_LWIP_TCP_QUEUE_OOSEQ 1 +#define CONFIG_LWIP_TCP_OOSEQ_TIMEOUT 6 +#define CONFIG_LWIP_TCP_OOSEQ_MAX_PBUFS 4 +#define CONFIG_LWIP_TCP_OVERSIZE_MSS 1 +#define CONFIG_LWIP_TCP_RTO_TIME 1500 +#define CONFIG_LWIP_MAX_UDP_PCBS 16 +#define CONFIG_LWIP_UDP_RECVMBOX_SIZE 6 +#define CONFIG_LWIP_CHECKSUM_CHECK_ICMP 1 +#define CONFIG_LWIP_TCPIP_TASK_STACK_SIZE 3072 +#define CONFIG_LWIP_TCPIP_TASK_AFFINITY_NO_AFFINITY 1 +#define CONFIG_LWIP_TCPIP_TASK_AFFINITY 0x7FFFFFFF +#define CONFIG_LWIP_IPV6_MEMP_NUM_ND6_QUEUE 3 +#define CONFIG_LWIP_IPV6_ND6_NUM_NEIGHBORS 5 +#define CONFIG_LWIP_IPV6_ND6_NUM_PREFIXES 5 +#define CONFIG_LWIP_IPV6_ND6_NUM_ROUTERS 3 +#define CONFIG_LWIP_IPV6_ND6_NUM_DESTINATIONS 10 +#define CONFIG_LWIP_ICMP 1 +#define CONFIG_LWIP_MAX_RAW_PCBS 16 +#define CONFIG_LWIP_SNTP_MAX_SERVERS 1 +#define CONFIG_LWIP_SNTP_UPDATE_DELAY 3600000 +#define CONFIG_LWIP_SNTP_STARTUP_DELAY 1 +#define CONFIG_LWIP_SNTP_MAXIMUM_STARTUP_DELAY 5000 +#define CONFIG_LWIP_DNS_MAX_HOST_IP 1 +#define CONFIG_LWIP_DNS_MAX_SERVERS 3 +#define CONFIG_LWIP_BRIDGEIF_MAX_PORTS 7 +#define CONFIG_LWIP_ESP_LWIP_ASSERT 1 +#define CONFIG_LWIP_HOOK_TCP_ISN_DEFAULT 1 +#define CONFIG_LWIP_HOOK_IP6_ROUTE_NONE 1 +#define CONFIG_LWIP_HOOK_ND6_GET_GW_NONE 1 +#define CONFIG_LWIP_HOOK_IP6_SELECT_SRC_ADDR_NONE 1 +#define CONFIG_LWIP_HOOK_DHCP_EXTRA_OPTION_NONE 1 +#define CONFIG_LWIP_HOOK_NETCONN_EXT_RESOLVE_NONE 1 +#define CONFIG_LWIP_HOOK_DNS_EXT_RESOLVE_NONE 1 +#define CONFIG_LWIP_HOOK_IP6_INPUT_DEFAULT 1 +#define CONFIG_MBEDTLS_VER_4_X_SUPPORT 1 +#define CONFIG_MBEDTLS_COMPILER_OPTIMIZATION_SIZE 1 +#define CONFIG_MBEDTLS_FS_IO 1 +#define CONFIG_MBEDTLS_THREADING_C 1 +#define CONFIG_MBEDTLS_THREADING_PTHREAD 1 +#define CONFIG_MBEDTLS_ERROR_STRINGS 1 +#define CONFIG_MBEDTLS_VERSION_C 1 +#define CONFIG_MBEDTLS_HAVE_TIME 1 +#define CONFIG_MBEDTLS_INTERNAL_MEM_ALLOC 1 +#define CONFIG_MBEDTLS_ASYMMETRIC_CONTENT_LEN 1 +#define CONFIG_MBEDTLS_SSL_IN_CONTENT_LEN 16384 +#define CONFIG_MBEDTLS_SSL_OUT_CONTENT_LEN 4096 +#define CONFIG_MBEDTLS_SELF_TEST 1 +#define CONFIG_MBEDTLS_X509_USE_C 1 +#define CONFIG_MBEDTLS_PEM_PARSE_C 1 +#define CONFIG_MBEDTLS_PEM_WRITE_C 1 +#define CONFIG_MBEDTLS_PK_C 1 +#define CONFIG_MBEDTLS_PK_PARSE_C 1 +#define CONFIG_MBEDTLS_PK_WRITE_C 1 +#define CONFIG_MBEDTLS_X509_CRL_PARSE_C 1 +#define CONFIG_MBEDTLS_X509_CRT_PARSE_C 1 +#define CONFIG_MBEDTLS_X509_CSR_PARSE_C 1 +#define CONFIG_MBEDTLS_X509_RSASSA_PSS_SUPPORT 1 +#define CONFIG_MBEDTLS_ASN1_PARSE_C 1 +#define CONFIG_MBEDTLS_ASN1_WRITE_C 1 +#define CONFIG_MBEDTLS_CERTIFICATE_BUNDLE 1 +#define CONFIG_MBEDTLS_CERTIFICATE_BUNDLE_DEFAULT_FULL 1 +#define CONFIG_MBEDTLS_CERTIFICATE_BUNDLE_MAX_CERTS 200 +#define CONFIG_MBEDTLS_TLS_ENABLED 1 +#define CONFIG_MBEDTLS_SSL_PROTO_TLS1_2 1 +#define CONFIG_MBEDTLS_TLS_SERVER 1 +#define CONFIG_MBEDTLS_TLS_CLIENT 1 +#define CONFIG_MBEDTLS_TLS_SERVER_AND_CLIENT 1 +#define CONFIG_MBEDTLS_SSL_CACHE_C 1 +#define CONFIG_MBEDTLS_SSL_ALL_ALERT_MESSAGES 1 +#define CONFIG_MBEDTLS_KEY_EXCHANGE_RSA 1 +#define CONFIG_MBEDTLS_KEY_EXCHANGE_ELLIPTIC_CURVE 1 +#define CONFIG_MBEDTLS_KEY_EXCHANGE_ECDHE_RSA 1 +#define CONFIG_MBEDTLS_KEY_EXCHANGE_ECDHE_ECDSA 1 +#define CONFIG_MBEDTLS_SSL_SERVER_NAME_INDICATION 1 +#define CONFIG_MBEDTLS_SSL_ALPN 1 +#define CONFIG_MBEDTLS_SSL_MAX_FRAGMENT_LENGTH 1 +#define CONFIG_MBEDTLS_SSL_RENEGOTIATION 1 +#define CONFIG_MBEDTLS_CLIENT_SSL_SESSION_TICKETS 1 +#define CONFIG_MBEDTLS_SERVER_SSL_SESSION_TICKETS 1 +#define CONFIG_MBEDTLS_AES_C 1 +#define CONFIG_MBEDTLS_CCM_C 1 +#define CONFIG_MBEDTLS_CIPHER_MODE_CBC 1 +#define CONFIG_MBEDTLS_CIPHER_MODE_CFB 1 +#define CONFIG_MBEDTLS_CIPHER_MODE_CTR 1 +#define CONFIG_MBEDTLS_CIPHER_MODE_OFB 1 +#define CONFIG_MBEDTLS_CIPHER_MODE_XTS 1 +#define CONFIG_MBEDTLS_GCM_C 1 +#define CONFIG_MBEDTLS_AES_ROM_TABLES 1 +#define CONFIG_MBEDTLS_CMAC_C 1 +#define CONFIG_MBEDTLS_RSA_C 1 +#define CONFIG_MBEDTLS_ECP_C 1 +#define CONFIG_MBEDTLS_ECP_DP_SECP256R1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_SECP384R1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_SECP521R1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_SECP256K1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_BP256R1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_BP384R1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_BP512R1_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_DP_CURVE25519_ENABLED 1 +#define CONFIG_MBEDTLS_ECP_NIST_OPTIM 1 +#define CONFIG_MBEDTLS_ECDH_C 1 +#define CONFIG_MBEDTLS_ECDSA_C 1 +#define CONFIG_MBEDTLS_PK_PARSE_EC_EXTENDED 1 +#define CONFIG_MBEDTLS_PK_PARSE_EC_COMPRESSED 1 +#define CONFIG_MBEDTLS_ECDSA_DETERMINISTIC 1 +#define CONFIG_MBEDTLS_MD_C 1 +#define CONFIG_MBEDTLS_MD5_C 1 +#define CONFIG_MBEDTLS_SHA1_C 1 +#define CONFIG_MBEDTLS_SHA256_C 1 +#define CONFIG_MBEDTLS_SHA384_C 1 +#define CONFIG_MBEDTLS_SHA512_C 1 +#define CONFIG_MBEDTLS_ROM_MD5 1 +#define CONFIG_MBEDTLS_HARDWARE_ECDSA_VERIFY 1 +#define CONFIG_MBEDTLS_HARDWARE_ECC 1 +#define CONFIG_MBEDTLS_ECC_OTHER_CURVES_SOFT_FALLBACK 1 +#define CONFIG_MBEDTLS_HARDWARE_SHA 1 +#define CONFIG_MBEDTLS_HARDWARE_MPI 1 +#define CONFIG_MBEDTLS_LARGE_KEY_SOFTWARE_MPI 1 +#define CONFIG_MBEDTLS_MPI_USE_INTERRUPT 1 +#define CONFIG_MBEDTLS_MPI_INTERRUPT_LEVEL 0 +#define CONFIG_MBEDTLS_HARDWARE_AES 1 +#define CONFIG_MBEDTLS_HARDWARE_GCM 1 +#define CONFIG_MBEDTLS_GCM_SUPPORT_NON_AES_CIPHER 1 +#define CONFIG_MBEDTLS_AES_USE_INTERRUPT 1 +#define CONFIG_MBEDTLS_AES_INTERRUPT_LEVEL 0 +#define CONFIG_MBEDTLS_AES_HW_SMALL_DATA_LEN_OPTIM 1 +#define CONFIG_MBEDTLS_HARDWARE_RSA_DS_PERIPHERAL 1 +#define CONFIG_MBEDTLS_CTR_DRBG_C 1 +#define CONFIG_MBEDTLS_HMAC_DRBG_C 1 +#define CONFIG_MBEDTLS_BASE64_C 1 +#define CONFIG_MBEDTLS_PKCS5_C 1 +#define CONFIG_MBEDTLS_PKCS7_C 1 +#define CONFIG_MBEDTLS_PKCS1_V15 1 +#define CONFIG_MBEDTLS_PKCS1_V21 1 +#define CONFIG_ESP_PROTOCOMM_SUPPORT_SECURITY_VERSION_2 1 +#define CONFIG_ESP_PROTOCOMM_SUPPORT_SECURITY_PATCH_VERSION 1 +#define CONFIG_PTHREAD_TASK_PRIO_DEFAULT 5 +#define CONFIG_PTHREAD_TASK_STACK_SIZE_DEFAULT 3072 +#define CONFIG_PTHREAD_STACK_MIN 768 +#define CONFIG_PTHREAD_DEFAULT_CORE_NO_AFFINITY 1 +#define CONFIG_PTHREAD_TASK_CORE_DEFAULT -1 +#define CONFIG_PTHREAD_TASK_NAME_DEFAULT "pthread" +#define CONFIG_SD_ENABLE_SDIO_SUPPORT 1 +#define CONFIG_MMU_PAGE_SIZE_64KB 1 +#define CONFIG_MMU_PAGE_MODE "64KB" +#define CONFIG_MMU_PAGE_SIZE 0x10000 +#define CONFIG_SPI_FLASH_BROWNOUT_RESET_XMC 1 +#define CONFIG_SPI_FLASH_BROWNOUT_RESET 1 +#define CONFIG_SPI_FLASH_HPM_AUTO 1 +#define CONFIG_SPI_FLASH_HPM_ON 1 +#define CONFIG_SPI_FLASH_HPM_DC_AUTO 1 +#define CONFIG_SPI_FLASH_SUSPEND_TSUS_VAL_US 50 +#define CONFIG_SPI_FLASH_PLACE_FUNCTIONS_IN_IRAM 1 +#define CONFIG_SPI_FLASH_DANGEROUS_WRITE_ABORTS 1 +#define CONFIG_SPI_FLASH_YIELD_DURING_ERASE 1 +#define CONFIG_SPI_FLASH_ERASE_YIELD_DURATION_MS 20 +#define CONFIG_SPI_FLASH_ERASE_YIELD_TICKS 1 +#define CONFIG_SPI_FLASH_WRITE_CHUNK_SIZE 8192 +#define CONFIG_SPI_FLASH_VENDOR_XMC_SUPPORT_ENABLED 1 +#define CONFIG_SPI_FLASH_VENDOR_GD_SUPPORT_ENABLED 1 +#define CONFIG_SPI_FLASH_SUPPORT_GD_CHIP 1 +#define CONFIG_SPI_FLASH_SUPPORT_BOYA_CHIP 1 +#define CONFIG_SPI_FLASH_ENABLE_ENCRYPTED_READ_WRITE 1 +#define CONFIG_SPIFFS_MAX_PARTITIONS 3 +#define CONFIG_SPIFFS_CACHE 1 +#define CONFIG_SPIFFS_CACHE_WR 1 +#define CONFIG_SPIFFS_PAGE_CHECK 1 +#define CONFIG_SPIFFS_GC_MAX_RUNS 10 +#define CONFIG_SPIFFS_PAGE_SIZE 256 +#define CONFIG_SPIFFS_OBJ_NAME_LEN 32 +#define CONFIG_SPIFFS_USE_MAGIC 1 +#define CONFIG_SPIFFS_USE_MAGIC_LENGTH 1 +#define CONFIG_SPIFFS_META_LENGTH 4 +#define CONFIG_SPIFFS_USE_MTIME 1 +#define CONFIG_WS_TRANSPORT 1 +#define CONFIG_WS_BUFFER_SIZE 1024 +#define CONFIG_UNITY_ENABLE_FLOAT 1 +#define CONFIG_UNITY_ENABLE_DOUBLE 1 +#define CONFIG_UNITY_ENABLE_IDF_TEST_RUNNER 1 +#define CONFIG_VFS_SUPPORT_IO 1 +#define CONFIG_VFS_SUPPORT_DIR 1 +#define CONFIG_VFS_SUPPORT_SELECT 1 +#define CONFIG_VFS_SUPPRESS_SELECT_DEBUG_OUTPUT 1 +#define CONFIG_VFS_MAX_COUNT 8 +#define CONFIG_VFS_SEMIHOSTFS_MAX_MOUNT_POINTS 1 +#define CONFIG_VFS_INITIALIZE_DEV_NULL 1 +#define CONFIG_WL_SECTOR_SIZE_4096 1 +#define CONFIG_WL_SECTOR_SIZE 4096 +#define CONFIG_EPPP_LINK_DEVICE_UART 1 +#define CONFIG_EPPP_LINK_CONN_MAX_RETRY 6 +#define CONFIG_ESP_HOSTED_ENABLED 1 +#define CONFIG_ESP_HOSTED_CP_TARGET_ESP32C6 1 +#define CONFIG_ESP_HOSTED_PRIV_ENABLE_WIFI_OPTIONS 1 +#define CONFIG_ESP_HOSTED_IDF_SLAVE_TARGET "esp32c6" +#define CONFIG_ESP_HOSTED_P4_DEV_BOARD_NONE 1 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_OPTION 1 +#define CONFIG_ESP_HOSTED_PRIV_SPI_HD_OPTION 1 +#define CONFIG_ESP_HOSTED_SDIO_HOST_INTERFACE 1 +#define CONFIG_ESP_HOSTED_SDIO_RESET_ACTIVE_HIGH 1 +#define CONFIG_ESP_HOSTED_SDIO_OPTIMIZATION_RX_STREAMING_MODE 1 +#define CONFIG_ESP_HOSTED_SDIO_SLOT_1 1 +#define CONFIG_ESP_HOSTED_SDIO_SLOT 1 +#define CONFIG_ESP_HOSTED_SDIO_4_BIT_BUS 1 +#define CONFIG_ESP_HOSTED_SDIO_BUS_WIDTH 4 +#define CONFIG_ESP_HOSTED_SDIO_CLOCK_FREQ_KHZ 40000 +#define CONFIG_ESP_HOSTED_SDIO_CMD_GPIO_RANGE_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_CMD_GPIO_RANGE_MAX 100 +#define CONFIG_ESP_HOSTED_SDIO_CLK_GPIO_RANGE_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_CLK_GPIO_RANGE_MAX 100 +#define CONFIG_ESP_HOSTED_SDIO_D0_GPIO_RANGE_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_D0_GPIO_RANGE_MAX 100 +#define CONFIG_ESP_HOSTED_SDIO_D1_GPIO_RANGE_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_D1_GPIO_RANGE_MAX 100 +#define CONFIG_ESP_HOSTED_SDIO_D2_GPIO_RANGE_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_D2_GPIO_RANGE_MAX 100 +#define CONFIG_ESP_HOSTED_SDIO_D3_GPIO_RANGE_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_D3_GPIO_RANGE_MAX 100 +#define CONFIG_ESP_HOSTED_SDIO_RESET_SLAVE_GPIO_MIN 0 +#define CONFIG_ESP_HOSTED_SDIO_RESET_SLAVE_GPIO_MAX 100 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_PIN_CMD_SLOT_1 19 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_PIN_CLK_SLOT_1 18 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_PIN_D0_SLOT_1 14 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_PIN_D1_4BIT_BUS_SLOT_1 15 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_PIN_D2_4BIT_BUS_SLOT_1 16 +#define CONFIG_ESP_HOSTED_PRIV_SDIO_PIN_D3_4BIT_BUS_SLOT_1 17 +#define CONFIG_ESP_HOSTED_SDIO_GPIO_RESET_SLAVE 54 +#define CONFIG_ESP_HOSTED_SDIO_PIN_CMD 19 +#define CONFIG_ESP_HOSTED_SDIO_PIN_CLK 18 +#define CONFIG_ESP_HOSTED_SDIO_PIN_D0 14 +#define CONFIG_ESP_HOSTED_SDIO_PRIV_PIN_D1_4BIT_BUS 15 +#define CONFIG_ESP_HOSTED_SDIO_PIN_D2 16 +#define CONFIG_ESP_HOSTED_SDIO_PIN_D3 17 +#define CONFIG_ESP_HOSTED_SDIO_PIN_D1 15 +#define CONFIG_ESP_HOSTED_SDIO_TX_Q_SIZE 20 +#define CONFIG_ESP_HOSTED_SDIO_RX_Q_SIZE 20 +#define CONFIG_ESP_HOSTED_SDIO_RESET_DELAY_MS 1500 +#define CONFIG_ESP_HOSTED_SLAVE_RESET_ON_EVERY_HOST_BOOTUP 1 +#define CONFIG_ESP_HOSTED_GPIO_SLAVE_RESET_SLAVE 54 +#define CONFIG_ESP_HOSTED_ENABLE_BT_NIMBLE 1 +#define CONFIG_ESP_HOSTED_NIMBLE_HCI_VHCI 1 +#define CONFIG_ESP_HOSTED_RPC_TASK_STACK 4096 +#define CONFIG_ESP_HOSTED_DFLT_TASK_STACK 3072 +#define CONFIG_ESP_HOSTED_TRANSPORT_RESTART_ON_FAILURE 1 +#define CONFIG_ESP_HOSTED_MEM_MONITOR 1 +#define CONFIG_ESP_HOSTED_ENABLE_ITWT 1 +#define CONFIG_ESP_HOSTED_USE_MEMPOOL 1 +#define CONFIG_ESP_HOSTED_MAX_SIMULTANEOUS_SYNC_RPC_REQUESTS 5 +#define CONFIG_ESP_HOSTED_MAX_SIMULTANEOUS_ASYNC_RPC_REQUESTS 5 +#define CONFIG_ESP_HOSTED_CLI_ENABLED 1 +#define CONFIG_ESP_HOSTED_HOST_TO_ESP_WIFI_DATA_THROTTLE 1 +#define CONFIG_ESP_HOSTED_PRIV_WIFI_TX_SDIO_HIGH_THRESHOLD 80 +#define CONFIG_ESP_HOSTED_TO_WIFI_DATA_THROTTLE_HIGH_THRESHOLD 80 +#define CONFIG_ESP_HOSTED_TO_WIFI_DATA_THROTTLE_LOW_THRESHOLD 60 +#define CONFIG_ESP_HOSTED_ENABLE_PEER_DATA_TRANSFER 1 +#define CONFIG_ESP_HOSTED_MAX_CUSTOM_MSG_HANDLERS 3 +#define CONFIG_ESP_WIFI_REMOTE_ENABLED 1 +#define CONFIG_ESP_WIFI_REMOTE_IDF_SPECIFIC_ADDED 1 +#define CONFIG_SLAVE_IDF_TARGET_ESP32C6 1 +#define CONFIG_SLAVE_SOC_WIFI_SUPPORTED 1 +#define CONFIG_SLAVE_SOC_WIFI_WAPI_SUPPORT 1 +#define CONFIG_SLAVE_SOC_WIFI_CSI_SUPPORT 1 +#define CONFIG_SLAVE_SOC_WIFI_MESH_SUPPORT 1 +#define CONFIG_SLAVE_SOC_WIFI_LIGHT_SLEEP_CLK_WIDTH 12 +#define CONFIG_SLAVE_SOC_WIFI_HW_TSF 1 +#define CONFIG_SLAVE_SOC_WIFI_FTM_SUPPORT 1 +#define CONFIG_SLAVE_FREERTOS_UNICORE 1 +#define CONFIG_SLAVE_SOC_WIFI_GCMP_SUPPORT 1 +#define CONFIG_SLAVE_SOC_WIFI_TXOP_SUPPORT 1 +#define CONFIG_SLAVE_IDF_TARGET_ARCH_RISCV 1 +#define CONFIG_SLAVE_SOC_WIFI_HE_SUPPORT 1 +#define CONFIG_SLAVE_SOC_WIFI_MAC_VERSION_NUM 2 +#define CONFIG_WIFI_RMT_STATIC_RX_BUFFER_NUM 10 +#define CONFIG_WIFI_RMT_DYNAMIC_RX_BUFFER_NUM 32 +#define CONFIG_WIFI_RMT_DYNAMIC_TX_BUFFER 1 +#define CONFIG_WIFI_RMT_TX_BUFFER_TYPE 1 +#define CONFIG_WIFI_RMT_DYNAMIC_TX_BUFFER_NUM 32 +#define CONFIG_WIFI_RMT_STATIC_RX_MGMT_BUFFER 1 +#define CONFIG_WIFI_RMT_DYNAMIC_RX_MGMT_BUF 0 +#define CONFIG_WIFI_RMT_RX_MGMT_BUF_NUM_DEF 5 +#define CONFIG_WIFI_RMT_AMPDU_TX_ENABLED 1 +#define CONFIG_WIFI_RMT_TX_BA_WIN 6 +#define CONFIG_WIFI_RMT_AMPDU_RX_ENABLED 1 +#define CONFIG_WIFI_RMT_RX_BA_WIN 6 +#define CONFIG_WIFI_RMT_NVS_ENABLED 1 +#define CONFIG_WIFI_RMT_SOFTAP_BEACON_MAX_LEN 752 +#define CONFIG_WIFI_RMT_MGMT_SBUF_NUM 32 +#define CONFIG_WIFI_RMT_IRAM_OPT 1 +#define CONFIG_WIFI_RMT_EXTRA_IRAM_OPT 1 +#define CONFIG_WIFI_RMT_RX_IRAM_OPT 1 +#define CONFIG_WIFI_RMT_ENABLE_WPA3_SAE 1 +#define CONFIG_WIFI_RMT_ENABLE_SAE_H2E 1 +#define CONFIG_WIFI_RMT_ENABLE_SAE_PK 1 +#define CONFIG_WIFI_RMT_SOFTAP_SAE_SUPPORT 1 +#define CONFIG_WIFI_RMT_ENABLE_WPA3_OWE_STA 1 +#define CONFIG_WIFI_RMT_WPA3_COMPATIBLE_SUPPORT 1 +#define CONFIG_WIFI_RMT_SLP_IRAM_OPT 1 +#define CONFIG_WIFI_RMT_SLP_DEFAULT_MIN_ACTIVE_TIME 50 +#define CONFIG_WIFI_RMT_BSS_MAX_IDLE_SUPPORT 1 +#define CONFIG_WIFI_RMT_SLP_DEFAULT_MAX_ACTIVE_TIME 10 +#define CONFIG_WIFI_RMT_SLP_DEFAULT_WAIT_BROADCAST_DATA_TIME 15 +#define CONFIG_WIFI_RMT_STA_DISCONNECTED_PM_ENABLE 1 +#define CONFIG_WIFI_RMT_GMAC_SUPPORT 1 +#define CONFIG_WIFI_RMT_SOFTAP_SUPPORT 1 +#define CONFIG_WIFI_RMT_ESPNOW_MAX_ENCRYPT_NUM 7 +#define CONFIG_WIFI_RMT_MBEDTLS_CRYPTO 1 +#define CONFIG_WIFI_RMT_MBEDTLS_TLS_CLIENT 1 +#define CONFIG_WIFI_RMT_TX_HETB_QUEUE_NUM 3 +#define CONFIG_WIFI_RMT_ENTERPRISE_SUPPORT 1 +#define CONFIG_ESP_WIFI_REMOTE_LIBRARY_HOSTED 1 +#define CONFIG_ESP_WIFI_REMOTE_EAP_ENABLED 1 + +/* List of deprecated options */ +#define CONFIG_BROWNOUT_DET CONFIG_ESP_BROWNOUT_DET +#define CONFIG_BROWNOUT_DET_LVL CONFIG_ESP_BROWNOUT_DET_LVL +#define CONFIG_BROWNOUT_DET_LVL_SEL_7 CONFIG_ESP_BROWNOUT_DET_LVL_SEL_7 +#define CONFIG_BT_NIMBLE_ACL_BUF_COUNT CONFIG_BT_NIMBLE_TRANSPORT_ACL_FROM_LL_COUNT +#define CONFIG_BT_NIMBLE_ACL_BUF_SIZE CONFIG_BT_NIMBLE_TRANSPORT_ACL_SIZE +#define CONFIG_BT_NIMBLE_HCI_EVT_BUF_SIZE CONFIG_BT_NIMBLE_TRANSPORT_EVT_SIZE +#define CONFIG_BT_NIMBLE_HCI_EVT_HI_BUF_COUNT CONFIG_BT_NIMBLE_TRANSPORT_EVT_COUNT +#define CONFIG_BT_NIMBLE_HCI_EVT_LO_BUF_COUNT CONFIG_BT_NIMBLE_TRANSPORT_EVT_DISCARD_COUNT +#define CONFIG_BT_NIMBLE_MSYS1_BLOCK_COUNT CONFIG_BT_NIMBLE_MSYS_1_BLOCK_COUNT +#define CONFIG_BT_NIMBLE_SM_SC_LVL CONFIG_BT_NIMBLE_SM_LVL +#define CONFIG_BT_NIMBLE_TASK_STACK_SIZE CONFIG_BT_NIMBLE_HOST_TASK_STACK_SIZE +#define CONFIG_COMPILER_OPTIMIZATION_DEFAULT CONFIG_COMPILER_OPTIMIZATION_DEBUG +#define CONFIG_COMPILER_OPTIMIZATION_LEVEL_DEBUG CONFIG_COMPILER_OPTIMIZATION_DEBUG +#define CONFIG_CONSOLE_UART CONFIG_ESP_CONSOLE_UART +#define CONFIG_CONSOLE_UART_BAUDRATE CONFIG_ESP_CONSOLE_UART_BAUDRATE +#define CONFIG_CONSOLE_UART_DEFAULT CONFIG_ESP_CONSOLE_UART_DEFAULT +#define CONFIG_CONSOLE_UART_NUM CONFIG_ESP_CONSOLE_UART_NUM +#define CONFIG_ESP32_DEFAULT_PTHREAD_CORE_NO_AFFINITY CONFIG_PTHREAD_DEFAULT_CORE_NO_AFFINITY +#define CONFIG_ESP32_ENABLE_COREDUMP_TO_NONE CONFIG_ESP_COREDUMP_ENABLE_TO_NONE +#define CONFIG_ESP32_PTHREAD_STACK_MIN CONFIG_PTHREAD_STACK_MIN +#define CONFIG_ESP32_PTHREAD_TASK_CORE_DEFAULT CONFIG_PTHREAD_TASK_CORE_DEFAULT +#define CONFIG_ESP32_PTHREAD_TASK_NAME_DEFAULT CONFIG_PTHREAD_TASK_NAME_DEFAULT +#define CONFIG_ESP32_PTHREAD_TASK_PRIO_DEFAULT CONFIG_PTHREAD_TASK_PRIO_DEFAULT +#define CONFIG_ESP32_PTHREAD_TASK_STACK_SIZE_DEFAULT CONFIG_PTHREAD_TASK_STACK_SIZE_DEFAULT +#define CONFIG_ESP32_WIFI_AMPDU_RX_ENABLED CONFIG_ESP_WIFI_AMPDU_RX_ENABLED +#define CONFIG_ESP32_WIFI_AMPDU_TX_ENABLED CONFIG_ESP_WIFI_AMPDU_TX_ENABLED +#define CONFIG_ESP32_WIFI_DYNAMIC_RX_BUFFER_NUM CONFIG_ESP_WIFI_DYNAMIC_RX_BUFFER_NUM +#define CONFIG_ESP32_WIFI_DYNAMIC_TX_BUFFER_NUM CONFIG_ESP_WIFI_DYNAMIC_TX_BUFFER_NUM +#define CONFIG_ESP32_WIFI_ENABLE_WPA3_OWE_STA CONFIG_ESP_WIFI_ENABLE_WPA3_OWE_STA +#define CONFIG_ESP32_WIFI_ENABLE_WPA3_SAE CONFIG_ESP_WIFI_ENABLE_WPA3_SAE +#define CONFIG_ESP32_WIFI_IRAM_OPT CONFIG_ESP_WIFI_IRAM_OPT +#define CONFIG_ESP32_WIFI_MGMT_SBUF_NUM CONFIG_ESP_WIFI_MGMT_SBUF_NUM +#define CONFIG_ESP32_WIFI_NVS_ENABLED CONFIG_ESP_WIFI_NVS_ENABLED +#define CONFIG_ESP32_WIFI_RX_BA_WIN CONFIG_ESP_WIFI_RX_BA_WIN +#define CONFIG_ESP32_WIFI_RX_IRAM_OPT CONFIG_ESP_WIFI_RX_IRAM_OPT +#define CONFIG_ESP32_WIFI_SOFTAP_BEACON_MAX_LEN CONFIG_ESP_WIFI_SOFTAP_BEACON_MAX_LEN +#define CONFIG_ESP32_WIFI_STATIC_RX_BUFFER_NUM CONFIG_ESP_WIFI_STATIC_RX_BUFFER_NUM +#define CONFIG_ESP32_WIFI_TX_BA_WIN CONFIG_ESP_WIFI_TX_BA_WIN +#define CONFIG_ESP32_WIFI_TX_BUFFER_TYPE CONFIG_ESP_WIFI_TX_BUFFER_TYPE +#define CONFIG_ESP_DFLT_TASK_STACK CONFIG_ESP_HOSTED_DFLT_TASK_STACK +#define CONFIG_ESP_ENABLE_BT_NIMBLE CONFIG_ESP_HOSTED_ENABLE_BT_NIMBLE +#define CONFIG_ESP_GPIO_SLAVE_RESET_SLAVE CONFIG_ESP_HOSTED_GPIO_SLAVE_RESET_SLAVE +#define CONFIG_ESP_GRATUITOUS_ARP CONFIG_LWIP_ESP_GRATUITOUS_ARP +#define CONFIG_ESP_MAX_SIMULTANEOUS_ASYNC_RPC_REQUESTS CONFIG_ESP_HOSTED_MAX_SIMULTANEOUS_ASYNC_RPC_REQUESTS +#define CONFIG_ESP_MAX_SIMULTANEOUS_SYNC_RPC_REQUESTS CONFIG_ESP_HOSTED_MAX_SIMULTANEOUS_SYNC_RPC_REQUESTS +#define CONFIG_ESP_NIMBLE_HCI_VHCI CONFIG_ESP_HOSTED_NIMBLE_HCI_VHCI +#define CONFIG_ESP_RPC_TASK_STACK CONFIG_ESP_HOSTED_RPC_TASK_STACK +#define CONFIG_ESP_SDIO_4_BIT_BUS CONFIG_ESP_HOSTED_SDIO_4_BIT_BUS +#define CONFIG_ESP_SDIO_BUS_WIDTH CONFIG_ESP_HOSTED_SDIO_BUS_WIDTH +#define CONFIG_ESP_SDIO_CLOCK_FREQ_KHZ CONFIG_ESP_HOSTED_SDIO_CLOCK_FREQ_KHZ +#define CONFIG_ESP_SDIO_GPIO_RESET_SLAVE CONFIG_ESP_HOSTED_SDIO_GPIO_RESET_SLAVE +#define CONFIG_ESP_SDIO_HOST_INTERFACE CONFIG_ESP_HOSTED_SDIO_HOST_INTERFACE +#define CONFIG_ESP_SDIO_OPTIMIZATION_RX_STREAMING_MODE CONFIG_ESP_HOSTED_SDIO_OPTIMIZATION_RX_STREAMING_MODE +#define CONFIG_ESP_SDIO_PIN_CLK CONFIG_ESP_HOSTED_SDIO_PIN_CLK +#define CONFIG_ESP_SDIO_PIN_CMD CONFIG_ESP_HOSTED_SDIO_PIN_CMD +#define CONFIG_ESP_SDIO_PIN_D0 CONFIG_ESP_HOSTED_SDIO_PIN_D0 +#define CONFIG_ESP_SDIO_PIN_D1 CONFIG_ESP_HOSTED_SDIO_PIN_D1 +#define CONFIG_ESP_SDIO_PIN_D2 CONFIG_ESP_HOSTED_SDIO_PIN_D2 +#define CONFIG_ESP_SDIO_PIN_D3 CONFIG_ESP_HOSTED_SDIO_PIN_D3 +#define CONFIG_ESP_SDIO_RX_Q_SIZE CONFIG_ESP_HOSTED_SDIO_RX_Q_SIZE +#define CONFIG_ESP_SDIO_TX_Q_SIZE CONFIG_ESP_HOSTED_SDIO_TX_Q_SIZE +#define CONFIG_ESP_SYSTEM_BROWNOUT_INTR CONFIG_ESP_BROWNOUT_USE_INTR +#define CONFIG_ESP_SYSTEM_MEMPROT_FEATURE CONFIG_ESP_SYSTEM_MEMPROT +#define CONFIG_ESP_SYSTEM_MEMPROT_FEATURE_VIA_TEE CONFIG_ESP_SYSTEM_MEMPROT +#define CONFIG_ESP_SYSTEM_PMP_IDRAM_SPLIT CONFIG_ESP_SYSTEM_MEMPROT +#define CONFIG_ESP_TASK_WDT CONFIG_ESP_TASK_WDT_INIT +#define CONFIG_ESP_USE_MEMPOOL CONFIG_ESP_HOSTED_USE_MEMPOOL +#define CONFIG_FLASHMODE_DIO CONFIG_ESPTOOLPY_FLASHMODE_DIO +#define CONFIG_GARP_TMR_INTERVAL CONFIG_LWIP_GARP_TMR_INTERVAL +#define CONFIG_GDBSTUB_MAX_TASKS CONFIG_ESP_GDBSTUB_MAX_TASKS +#define CONFIG_GDBSTUB_SUPPORT_TASKS CONFIG_ESP_GDBSTUB_SUPPORT_TASKS +#define CONFIG_HOST_TO_ESP_WIFI_DATA_THROTTLE CONFIG_ESP_HOSTED_HOST_TO_ESP_WIFI_DATA_THROTTLE +#define CONFIG_IDF_SLAVE_TARGET CONFIG_ESP_HOSTED_IDF_SLAVE_TARGET +#define CONFIG_INT_WDT CONFIG_ESP_INT_WDT +#define CONFIG_INT_WDT_CHECK_CPU1 CONFIG_ESP_INT_WDT_CHECK_CPU1 +#define CONFIG_INT_WDT_TIMEOUT_MS CONFIG_ESP_INT_WDT_TIMEOUT_MS +#define CONFIG_IPC_TASK_STACK_SIZE CONFIG_ESP_IPC_TASK_STACK_SIZE +#define CONFIG_LOG_BOOTLOADER_LEVEL CONFIG_BOOTLOADER_LOG_LEVEL +#define CONFIG_LOG_BOOTLOADER_LEVEL_INFO CONFIG_BOOTLOADER_LOG_LEVEL_INFO +#define CONFIG_MAIN_TASK_STACK_SIZE CONFIG_ESP_MAIN_TASK_STACK_SIZE +#define CONFIG_MONITOR_BAUD CONFIG_ESPTOOLPY_MONITOR_BAUD +#define CONFIG_NEWLIB_STDIN_LINE_ENDING_CR CONFIG_LIBC_STDIN_LINE_ENDING_CR +#define CONFIG_NEWLIB_STDOUT_LINE_ENDING_CRLF CONFIG_LIBC_STDOUT_LINE_ENDING_CRLF +#define CONFIG_NEWLIB_TIME_SYSCALL_USE_RTC_HRT CONFIG_LIBC_TIME_SYSCALL_USE_RTC_HRT +#define CONFIG_NIMBLE_ATT_PREFERRED_MTU CONFIG_BT_NIMBLE_ATT_PREFERRED_MTU +#define CONFIG_NIMBLE_ENABLED CONFIG_BT_NIMBLE_ENABLED +#define CONFIG_NIMBLE_GAP_DEVICE_NAME_MAX_LEN CONFIG_BT_NIMBLE_GAP_DEVICE_NAME_MAX_LEN +#define CONFIG_NIMBLE_L2CAP_COC_MAX_NUM CONFIG_BT_NIMBLE_L2CAP_COC_MAX_NUM +#define CONFIG_NIMBLE_MAX_BONDS CONFIG_BT_NIMBLE_MAX_BONDS +#define CONFIG_NIMBLE_MAX_CCCDS CONFIG_BT_NIMBLE_MAX_CCCDS +#define CONFIG_NIMBLE_MAX_CONNECTIONS CONFIG_BT_NIMBLE_MAX_CONNECTIONS +#define CONFIG_NIMBLE_MEM_ALLOC_MODE_INTERNAL CONFIG_BT_NIMBLE_MEM_ALLOC_MODE_INTERNAL +#define CONFIG_NIMBLE_PINNED_TO_CORE CONFIG_BT_NIMBLE_PINNED_TO_CORE +#define CONFIG_NIMBLE_PINNED_TO_CORE_0 CONFIG_BT_NIMBLE_PINNED_TO_CORE_0 +#define CONFIG_NIMBLE_ROLE_BROADCASTER CONFIG_BT_NIMBLE_ROLE_BROADCASTER +#define CONFIG_NIMBLE_ROLE_OBSERVER CONFIG_BT_NIMBLE_ROLE_OBSERVER +#define CONFIG_NIMBLE_ROLE_PERIPHERAL CONFIG_BT_NIMBLE_ROLE_PERIPHERAL +#define CONFIG_NIMBLE_RPA_TIMEOUT CONFIG_BT_NIMBLE_RPA_TIMEOUT +#define CONFIG_NIMBLE_SM_LEGACY CONFIG_BT_NIMBLE_SM_LEGACY +#define CONFIG_NIMBLE_SM_SC CONFIG_BT_NIMBLE_SM_SC +#define CONFIG_NIMBLE_SVC_GAP_APPEARANCE CONFIG_BT_NIMBLE_SVC_GAP_APPEARANCE +#define CONFIG_NIMBLE_SVC_GAP_DEVICE_NAME CONFIG_BT_NIMBLE_SVC_GAP_DEVICE_NAME +#define CONFIG_NIMBLE_TASK_STACK_SIZE CONFIG_BT_NIMBLE_HOST_TASK_STACK_SIZE +#define CONFIG_OPTIMIZATION_ASSERTIONS_ENABLED CONFIG_COMPILER_OPTIMIZATION_ASSERTIONS_ENABLE +#define CONFIG_OPTIMIZATION_ASSERTION_LEVEL CONFIG_COMPILER_OPTIMIZATION_ASSERTION_LEVEL +#define CONFIG_OPTIMIZATION_LEVEL_DEBUG CONFIG_COMPILER_OPTIMIZATION_DEBUG +#define CONFIG_PERIPH_CTRL_FUNC_IN_IRAM CONFIG_ESP_PERIPH_CTRL_FUNC_IN_IRAM +#define CONFIG_POST_EVENTS_FROM_IRAM_ISR CONFIG_ESP_EVENT_POST_FROM_IRAM_ISR +#define CONFIG_POST_EVENTS_FROM_ISR CONFIG_ESP_EVENT_POST_FROM_ISR +#define CONFIG_PRIV_WIFI_TX_SDIO_HIGH_THRESHOLD CONFIG_ESP_HOSTED_PRIV_WIFI_TX_SDIO_HIGH_THRESHOLD +#define CONFIG_SDIO_RESET_ACTIVE_HIGH CONFIG_ESP_HOSTED_SDIO_RESET_ACTIVE_HIGH +#define CONFIG_SEMIHOSTFS_MAX_MOUNT_POINTS CONFIG_VFS_SEMIHOSTFS_MAX_MOUNT_POINTS +#define CONFIG_SPIRAM_ALLOW_STACK_EXTERNAL_MEMORY CONFIG_FREERTOS_TASK_CREATE_ALLOW_EXT_MEM +#define CONFIG_SPI_FLASH_WRITING_DANGEROUS_REGIONS_ABORTS CONFIG_SPI_FLASH_DANGEROUS_WRITE_ABORTS +#define CONFIG_STACK_CHECK_NONE CONFIG_COMPILER_STACK_CHECK_MODE_NONE +#define CONFIG_SUPPRESS_SELECT_DEBUG_OUTPUT CONFIG_VFS_SUPPRESS_SELECT_DEBUG_OUTPUT +#define CONFIG_SYSTEM_EVENT_QUEUE_SIZE CONFIG_ESP_SYSTEM_EVENT_QUEUE_SIZE +#define CONFIG_SYSTEM_EVENT_TASK_STACK_SIZE CONFIG_ESP_SYSTEM_EVENT_TASK_STACK_SIZE +#define CONFIG_TASK_WDT CONFIG_ESP_TASK_WDT_INIT +#define CONFIG_TASK_WDT_CHECK_IDLE_TASK_CPU0 CONFIG_ESP_TASK_WDT_CHECK_IDLE_TASK_CPU0 +#define CONFIG_TASK_WDT_CHECK_IDLE_TASK_CPU1 CONFIG_ESP_TASK_WDT_CHECK_IDLE_TASK_CPU1 +#define CONFIG_TASK_WDT_TIMEOUT_S CONFIG_ESP_TASK_WDT_TIMEOUT_S +#define CONFIG_TCPIP_RECVMBOX_SIZE CONFIG_LWIP_TCPIP_RECVMBOX_SIZE +#define CONFIG_TCPIP_TASK_AFFINITY CONFIG_LWIP_TCPIP_TASK_AFFINITY +#define CONFIG_TCPIP_TASK_AFFINITY_NO_AFFINITY CONFIG_LWIP_TCPIP_TASK_AFFINITY_NO_AFFINITY +#define CONFIG_TCPIP_TASK_STACK_SIZE CONFIG_LWIP_TCPIP_TASK_STACK_SIZE +#define CONFIG_TCP_MAXRTX CONFIG_LWIP_TCP_MAXRTX +#define CONFIG_TCP_MSL CONFIG_LWIP_TCP_MSL +#define CONFIG_TCP_MSS CONFIG_LWIP_TCP_MSS +#define CONFIG_TCP_OVERSIZE_MSS CONFIG_LWIP_TCP_OVERSIZE_MSS +#define CONFIG_TCP_QUEUE_OOSEQ CONFIG_LWIP_TCP_QUEUE_OOSEQ +#define CONFIG_TCP_RECVMBOX_SIZE CONFIG_LWIP_TCP_RECVMBOX_SIZE +#define CONFIG_TCP_SND_BUF_DEFAULT CONFIG_LWIP_TCP_SND_BUF_DEFAULT +#define CONFIG_TCP_SYNMAXRTX CONFIG_LWIP_TCP_SYNMAXRTX +#define CONFIG_TCP_WND_DEFAULT CONFIG_LWIP_TCP_WND_DEFAULT +#define CONFIG_TIMER_QUEUE_LENGTH CONFIG_FREERTOS_TIMER_QUEUE_LENGTH +#define CONFIG_TIMER_TASK_PRIORITY CONFIG_FREERTOS_TIMER_TASK_PRIORITY +#define CONFIG_TIMER_TASK_STACK_DEPTH CONFIG_FREERTOS_TIMER_TASK_STACK_DEPTH +#define CONFIG_TIMER_TASK_STACK_SIZE CONFIG_ESP_TIMER_TASK_STACK_SIZE +#define CONFIG_TO_WIFI_DATA_THROTTLE_HIGH_THRESHOLD CONFIG_ESP_HOSTED_TO_WIFI_DATA_THROTTLE_HIGH_THRESHOLD +#define CONFIG_TO_WIFI_DATA_THROTTLE_LOW_THRESHOLD CONFIG_ESP_HOSTED_TO_WIFI_DATA_THROTTLE_LOW_THRESHOLD +#define CONFIG_UDP_RECVMBOX_SIZE CONFIG_LWIP_UDP_RECVMBOX_SIZE +#define CONFIG_WPA_MBEDTLS_CRYPTO CONFIG_ESP_WIFI_MBEDTLS_CRYPTO +#define CONFIG_WPA_MBEDTLS_TLS_CLIENT CONFIG_ESP_WIFI_MBEDTLS_TLS_CLIENT diff --git a/src/net/hosted/wifi_shim.c b/src/net/hosted/wifi_shim.c new file mode 100644 index 0000000..67bdcc6 --- /dev/null +++ b/src/net/hosted/wifi_shim.c @@ -0,0 +1,200 @@ +/* + * A narrow C surface over ESP-Hosted's Wi-Fi RPC, so Zig never transcribes an IDF struct. + * + * `rpc_wifi_init` takes a `wifi_init_config_t`, `rpc_wifi_set_config` takes a `wifi_config_t`, and + * scanning hands back `wifi_ap_record_t`. Those are large, versioned structs full of bitfields, and + * IDF builds them with macros - WIFI_INIT_CONFIG_DEFAULT() alone sets over twenty fields + * (esp_wifi.h:316). Writing Zig `extern struct`s to match would be a transcription that compiles + * happily and goes wrong on the next IDF release, exactly the mistake that + * `esp_hosted_sdio_get_config` taught this project once already. + * + * So the structs stay on the C side, built by IDF's own macros, and Zig gets plain scalars and byte + * buffers. Everything here is a thin forwarder; the interesting code is all in ESP-Hosted's RPC + * layer, which this does not duplicate. + */ + +#include <string.h> + +#include "esp_wifi_types.h" +#include "esp_wifi.h" +#include "rpc_wrap.h" + +/* ESP-Hosted's RPC entry points (host/drivers/rpc/wrap/rpc_wrap.c). Declared here rather than + * relying on the header, so a signature change is a compile error in this file. */ +int rpc_wifi_init(const wifi_init_config_t *arg); +int rpc_wifi_set_mode(wifi_mode_t mode); +int rpc_wifi_set_config(wifi_interface_t interface, wifi_config_t *conf); +int rpc_wifi_connect(void); +int rpc_wifi_scan_start(const wifi_scan_config_t *config, bool block); +int rpc_wifi_scan_get_ap_num(uint16_t *number); +int rpc_wifi_scan_get_ap_records(uint16_t *number, wifi_ap_record_t *ap_records); +int rpc_wifi_start(void); +int rpc_wifi_get_mac(wifi_interface_t mode, uint8_t mac[6]); +int rpc_wifi_set_ps(wifi_ps_type_t type); + +/* Initialise the coprocessor's Wi-Fi and put it in station mode, started. + * + * The order is IDF's own and is not negotiable: init, set_mode, start. `esp_wifi_start` is what + * actually brings the radio up on the C6; a config set before it is accepted and a connect before + * it is not. */ +int hosted_wifi_sta_start(void) +{ + /* IDF's WIFI_INIT_CONFIG_DEFAULT() is deliberately NOT used, and this is not a shortcut. + * + * That macro's first two fields are `.osi_funcs = &g_wifi_osi_funcs` and + * `.wpa_crypto_funcs = g_wifi_default_wpa_crypto_funcs` (esp_wifi.h:317-318) - the local Wi-Fi + * driver's OS adapter and crypto tables. This chip has no Wi-Fi driver: the C6 does, and it uses + * its own. Referencing them here pulls in symbols that cannot exist in this image, which is + * exactly the link error that led to this comment. + * + * They are also provably unnecessary. rpc_req.c:182-215 packs the request field by field, and + * every field it packs is a scalar; neither function pointer is ever serialised. So the struct + * only has to carry the scalars, and those come from the same Kconfig-derived macros the real + * default uses - via the checked-in sdkconfig, so they are this project's configuration and not + * a second set of numbers. + * + * `magic` is load-bearing: the coprocessor validates it (esp_wifi.h's own note says it must + * always be WIFI_INIT_CONFIG_MAGIC), so a zeroed struct is rejected. */ + wifi_init_config_t cfg = { 0 }; + cfg.static_rx_buf_num = CONFIG_ESP_WIFI_STATIC_RX_BUFFER_NUM; + cfg.dynamic_rx_buf_num = CONFIG_ESP_WIFI_DYNAMIC_RX_BUFFER_NUM; + cfg.tx_buf_type = CONFIG_ESP_WIFI_TX_BUFFER_TYPE; + cfg.static_tx_buf_num = WIFI_STATIC_TX_BUFFER_NUM; + cfg.dynamic_tx_buf_num = WIFI_DYNAMIC_TX_BUFFER_NUM; + cfg.rx_mgmt_buf_type = CONFIG_ESP_WIFI_DYNAMIC_RX_MGMT_BUF; + cfg.rx_mgmt_buf_num = WIFI_RX_MGMT_BUF_NUM_DEF; + cfg.cache_tx_buf_num = WIFI_CACHE_TX_BUFFER_NUM; + cfg.csi_enable = WIFI_CSI_ENABLED; + cfg.ampdu_rx_enable = WIFI_AMPDU_RX_ENABLED; + cfg.ampdu_tx_enable = WIFI_AMPDU_TX_ENABLED; + cfg.amsdu_tx_enable = WIFI_AMSDU_TX_ENABLED; + cfg.nvs_enable = WIFI_NVS_ENABLED; + cfg.nano_enable = WIFI_NANO_FORMAT_ENABLED; + cfg.rx_ba_win = WIFI_DEFAULT_RX_BA_WIN; + cfg.wifi_task_core_id = WIFI_TASK_CORE_ID; + cfg.beacon_max_len = WIFI_SOFTAP_BEACON_MAX_LEN; + cfg.mgmt_sbuf_num = WIFI_MGMT_SBUF_NUM; + cfg.feature_caps = WIFI_FEATURE_CAPS; + cfg.sta_disconnected_pm = WIFI_STA_DISCONNECTED_PM_ENABLED; + cfg.espnow_max_encrypt_num = CONFIG_ESP_WIFI_ESPNOW_MAX_ENCRYPT_NUM; + cfg.tx_hetb_queue_num = WIFI_TX_HETB_QUEUE_NUM; + cfg.dump_hesigb_enable = WIFI_DUMP_HESIGB_ENABLED; + cfg.magic = WIFI_INIT_CONFIG_MAGIC; + + int err = rpc_wifi_init(&cfg); + if (err) { + return err; + } + err = rpc_wifi_set_mode(WIFI_MODE_STA); + if (err) { + return err; + } + err = rpc_wifi_start(); + if (err) { + return err; + } + + /* Power save OFF, and this is not a performance tweak - it decides whether the board is + * reachable at all. + * + * ESP-IDF's default is WIFI_PS_MIN_MODEM (esp_wifi_types_generic.h:376): the station sleeps and + * only wakes for a beacon every DTIM period. A sleeping station misses frames the AP does not + * buffer for it, and broadcast ARP is exactly that. The observed symptom on this board was + * precise and misleading: DHCP completed - because the host speaks first and the reply arrives + * inside the wake window - the board took a real lease, and then it answered no ARP and no ping, + * with the frame counter advancing about once per ten seconds. It looked like a broken receive + * path rather than a radio that was asleep. + * + * A device that exists to answer requests cannot sleep between them. WIFI_PS_NONE. */ + return rpc_wifi_set_ps(WIFI_PS_NONE); +} + +/* The station's MAC. Needed by the IP stack: ARP and Ethernet framing are built around it, and it + * belongs to the C6's radio, not to this chip. */ +int hosted_wifi_get_mac(uint8_t out[6]) +{ + return rpc_wifi_get_mac(WIFI_IF_STA, out); +} + +/* Scan every channel and report how many networks were seen. + * + * Blocking: the RPC layer waits for the coprocessor to finish, which takes a couple of seconds + * across all channels. A scan needs no credentials, which makes it the cheapest end-to-end proof + * that the RPC path and the radio both work. */ +int hosted_wifi_scan(uint16_t *found) +{ + wifi_scan_config_t scan = { 0 }; + scan.show_hidden = true; + int err = rpc_wifi_scan_start(&scan, true); + if (err) { + return err; + } + return rpc_wifi_scan_get_ap_num(found); +} + +/* One scan result, flattened to scalars. + * + * `ssid_out` must have room for 33 bytes; the SSID is copied NUL-terminated. Returns the number of + * records actually written into the caller's view, which is `min(*count, what the slave has)`. + */ +int hosted_wifi_scan_record(uint16_t index, char *ssid_out, int8_t *rssi_out, + uint8_t *channel_out, uint8_t *authmode_out) +{ + /* One record at a time, into a local, so the caller never sees a wifi_ap_record_t. Asking the + * slave for a single record by index is not part of the RPC, so this fetches the run up to + * `index` and keeps the last - fine for the small numbers a diagnostic prints, and stated here + * rather than hidden because it is O(n^2) if someone loops it over hundreds of networks. */ + static wifi_ap_record_t records[16]; + uint16_t want = index + 1; + if (want > 16) { + return -1; + } + int err = rpc_wifi_scan_get_ap_records(&want, records); + if (err) { + return err; + } + if (index >= want) { + return -1; + } + + const wifi_ap_record_t *r = &records[index]; + size_t n = strnlen((const char *)r->ssid, 32); + memcpy(ssid_out, r->ssid, n); + ssid_out[n] = 0; + *rssi_out = r->rssi; + *channel_out = r->primary; + *authmode_out = (uint8_t)r->authmode; + return 0; +} + +/* Join a network. + * + * `ssid` and `psk` are NUL-terminated. The PSK is copied into the request and never stored here; + * it arrives from a build option so it is not in the source, and this function keeps no copy after + * the RPC returns. + * + * `threshold.authmode` is deliberately left at 0 (WIFI_AUTH_OPEN) rather than forced to WPA2: it is + * a *minimum* acceptable security level, and pinning it too high refuses networks that would + * otherwise work while pinning it low refuses nothing. The AP's actual authmode is what gets used. + */ +int hosted_wifi_connect(const char *ssid, const char *psk) +{ + wifi_config_t conf = { 0 }; + + size_t ssid_len = strnlen(ssid, sizeof(conf.sta.ssid) - 1); + memcpy(conf.sta.ssid, ssid, ssid_len); + + size_t psk_len = strnlen(psk, sizeof(conf.sta.password) - 1); + memcpy(conf.sta.password, psk, psk_len); + + /* Scan all channels and pick the strongest match rather than the first: this network has both a + * 2.4 GHz and a 5 GHz radio on the same SSID family, and the C6 is 2.4 GHz only. */ + conf.sta.scan_method = WIFI_ALL_CHANNEL_SCAN; + conf.sta.sort_method = WIFI_CONNECT_AP_BY_SIGNAL; + + int err = rpc_wifi_set_config(WIFI_IF_STA, &conf); + if (err) { + return err; + } + return rpc_wifi_connect(); +} diff --git a/src/net/hosted_glue.zig b/src/net/hosted_glue.zig new file mode 100644 index 0000000..40bac2e --- /dev/null +++ b/src/net/hosted_glue.zig @@ -0,0 +1,320 @@ +//! The symbols ESP-Hosted's transport needs that are neither libc nor the `g_h` port table: +//! this board's transport configuration, a logging sink, and honest stubs for the layers above +//! the transport that milestone 1 does not run. +//! +//! Measured, not guessed. Linking transport_drv.o + transport_util.o + sdio_drv.o + mempool.o +//! leaves 26 undefined symbols. src/net/libc.zig covers the libc ones, src/net/port.zig covers +//! `g_h`, and everything else is here. +//! +//! The distinction that matters in this file: a *configuration* symbol returns real values for this +//! board, and a *stub* prints its own name and parks. Nothing here silently returns success. On a +//! board with no debugger, a function that quietly does nothing is indistinguishable from a +//! function that worked, and that is the failure mode this project keeps paying for. + +const std = @import("std"); +const soc = @import("soc"); +const hal = @import("hal"); + +// --------------------------------------------------------------------------------------------- +// Board configuration is NOT here, deliberately. +// +// An earlier version of this file hand-wrote `esp_hosted_sdio_get_config` and +// `esp_hosted_transport_get_reset_config` in Zig, with a Zig `extern struct` mirroring +// `struct esp_hosted_sdio_config`. That was wrong twice over: the real signature takes a +// `struct esp_hosted_sdio_config **` and hands back a pointer to static storage rather than +// filling a caller's struct (host/api/include/esp_hosted_transport_config.h:151), and the real +// struct interleaves `gpio_pin_t {void *port; int pin;}` pairs rather than plain ints +// (same header, lines 22-45). A transcription of that layout is a silent wrong-pin bug waiting +// to happen. +// +// ESP-Hosted already ships both getters, deriving every value from Kconfig: +// host/api/src/esp_hosted_transport_config.c +// host/port/esp/freertos/src/port_esp_hosted_host_transport_defaults.c +// Together they compile clean under our flags and need only esp_log, esp_log_timestamp and the +// ROM's mem* - so the build compiles them instead. Nothing is transcribed and nothing can drift. +// +// What guarantees they produce THIS board's wiring is a compile-time check, not a comment: +// src/net/hosted/pin_assert.c static-asserts the Kconfig macros against the measured pin map +// (slot 1, 4-bit, 40 MHz, CLK 18, CMD 19, D0-D3 = 14/15/16/17, C6 reset 54) and is compiled as +// part of the hosted build. If a Kconfig value ever drifts from the board, the build fails with +// the name of the pin instead of the radio silently not answering. + +// --------------------------------------------------------------------------------------------- +// Logging +// +// ESP-Hosted logs through IDF's `esp_log`. Routing it to the ROM UART printer keeps the transport's +// own diagnostics - which are good, and are how we will see the handshake progress - without +// linking esp_log_write, its lock, its timestamp source or its level filtering. +// --------------------------------------------------------------------------------------------- + +/// IDF's log levels, from esp_log_level_t. +pub const Level = enum(c_int) { none = 0, err = 1, warn = 2, info = 3, debug = 4, verbose = 5 }; + +/// Everything at or below this prints. `.debug` while bringing the transport up: its per-packet +/// logging is the only view into the handshake before the IP stack exists. +var level: Level = .debug; + +pub fn setLevel(l: Level) void { + level = l; +} + +export fn esp_log_timestamp() callconv(.c) u32 { + // Milliseconds since boot, from the 16 MHz systimer - the only trustworthy timebase on this + // die, since the CPU runs at the bootloader's 90 MHz and nothing reconfigures the PLL. + // + // `read` is optional because the counter has a latch-then-read handshake that can fail to + // complete (see src/hal/systimer.zig, and the differential case that exists because of it). + // A failed read yields 0 rather than propagating: this is a log timestamp, and a logging call + // that panics would destroy exactly the diagnostics being printed. + const ticks = hal.systimer.read(.unit0) orelse return 0; + return @intCast(ticks / 16_000); +} + +export fn esp_log_level_get(_: ?[*:0]const u8) callconv(.c) c_int { + return @intFromEnum(level); +} + +export fn esp_log_level_set(_: ?[*:0]const u8, _: c_int) callconv(.c) void { + // Deliberately ignored: this build has one global level, set from Zig. Silently accepting the + // call is right here - a caller lowering a tag's verbosity is not load-bearing - and it is the + // only silent no-op in this file. +} + +export fn esp_log_default_level() callconv(.c) c_int { + return @intFromEnum(level); +} + +/// IDF v6's variadic log entry point. ESP-Hosted's ESP_LOG* macros land here. +export fn esp_log( + cfg: u32, + // Unused: the tag is already inside `fmt`, put there by ESP-Hosted's own logging macros. Kept + // in the signature because this is a C ABI entry point and the argument is really passed. + _: ?[*:0]const u8, + fmt: ?[*:0]const u8, + ..., +) callconv(.c) void { + // esp_log_config_t packs the level into the low bits; IDF's esp_log_level_t ordering means a + // numerically higher value is more verbose. + // The level lives in the low 3 bits: esp_log_config_t's `log_level` field is declared + // `esp_log_level_t log_level: ESP_LOG_LEVEL_LEN` (esp_log_config.h:122) with ESP_LOG_LEVEL_LEN + // = 3 (esp_log_level.h:30). Verified on the die by printing the raw word: info lines arrive as + // cfg=0x00000003 and warnings as cfg=0x00000002. + const msg_level: c_int = @intCast(cfg & 0x7); + if (msg_level > @intFromEnum(level)) return; + + var ap = @cVaStart(); + defer @cVaEnd(&ap); + emit(fmt orelse "", &ap); +} + +/// The other entry point in IDF v6; same job, already-started va_list. +export fn esp_log_writev( + msg_level: c_int, + _: ?[*:0]const u8, + fmt: ?[*:0]const u8, + ap: *std.builtin.VaList, +) callconv(.c) void { + if (msg_level > @intFromEnum(level)) return; + emit(fmt orelse "", ap); +} + +/// Format one already-level-filtered log line and put it on the wire. +/// +/// Takes neither the level nor the tag, and that is the point: ESP-Hosted's logging macros +/// (esp_hosted_log.h) bake the level letter, the timestamp and the tag into the format string they +/// hand us. An earlier version of this function added its own prefix as well, and every line came +/// out doubled: +/// +/// W (136) H_SDIO_DRV: W (136) H_SDIO_DRV: provided sdio tx queue size is zero! +/// +/// The level still matters - `esp_log` and `esp_log_writev` filter on it before calling here - it +/// just has no business in the output a second time. +fn emit(fmt: [*:0]const u8, ap: *std.builtin.VaList) void { + var line: [256]u8 = undefined; + const n = vsnprintf(&line, line.len, fmt, ap); + if (n <= 0) return; + const len = @min(@as(usize, @intCast(n)), line.len - 1); + // Print with plain `%s`, never `%.*s`: the ROM's `ets_printf` does not implement `.*` + // precision and prints the specifier literally, which is how the first run of this code + // produced "W (141) H_SDIO_DRV: %*0s" instead of a message. + line[len] = 0; + soc.rom.print("%s", .{@as([*:0]const u8, @ptrCast(&line))}); + // ESP-Hosted's own lines already end in \n; anything else gets a terminator so the next line + // does not run into it. + if (line[len - 1] != '\n') soc.rom.print("\r\n", .{}); +} + +extern fn vsnprintf(buf: [*]u8, size: usize, fmt: [*:0]const u8, ap: *std.builtin.VaList) c_int; + +/// Reached by protobuf-c's error paths. Lives here rather than in src/net/libc.zig because it needs +/// the ROM printer, and libc.zig is deliberately free of chip imports so it can be host-tested. +export fn printf(fmt: [*:0]const u8, ...) callconv(.c) c_int { + var ap = @cVaStart(); + defer @cVaEnd(&ap); + var line: [256]u8 = undefined; + const n = vsnprintf(&line, line.len, fmt, &ap); + if (n > 0) soc.rom.print("%s", .{@as([*:0]const u8, @ptrCast(&line))}); + return n; +} + +/// IDF's hex dump, referenced by ESP-Hosted's ESP_HEXLOG* macros once DEBUG-level logging is +/// compiled in (sdkconfig.h override 5). A real implementation, because a hexdump that prints +/// nothing is worse than none at all when the thing being debugged is a wire format - but bounded to +/// 64 bytes a call, since the point is to identify a packet rather than to transcribe it. +export fn esp_log_buffer_hexdump_internal( + tag: ?[*:0]const u8, + buffer: ?*const anyopaque, + buff_len: u16, + msg_level: c_int, +) callconv(.c) void { + if (msg_level > @intFromEnum(level)) return; + const bytes: [*]const u8 = @ptrCast(buffer orelse return); + const n = @min(buff_len, 64); + soc.rom.print("%s: %u bytes:", .{ @as([*:0]const u8, tag orelse "hex"), @as(u32, buff_len) }); + for (0..n) |i| soc.rom.print(" %02x", .{@as(u32, bytes[i])}); + if (n < buff_len) soc.rom.print(" ...", .{}); + soc.rom.print("\r\n", .{}); +} + +export fn esp_rom_printf(fmt: [*:0]const u8, ...) callconv(.c) c_int { + // The ROM printer is what this is named after; hand it straight over. + var ap = @cVaStart(); + defer @cVaEnd(&ap); + var line: [256]u8 = undefined; + const n = vsnprintf(&line, line.len, fmt, &ap); + if (n > 0) soc.rom.print("%s", .{@as([*:0]const u8, @ptrCast(&line))}); + return n; +} + +// --------------------------------------------------------------------------------------------- +// Event base +// +// `ESP_HOSTED_EVENT` and `WIFI_EVENT` are esp_event base symbols - opaque pointers whose *address* +// is the identity. Nothing dereferences them, so a byte of storage each is a complete +// implementation, and `_h_event_post` in port.zig is what actually routes events. +// --------------------------------------------------------------------------------------------- + +export const ESP_HOSTED_EVENT: u8 = 0; +export const WIFI_EVENT: u8 = 0; + +// --------------------------------------------------------------------------------------------- +// Stubs for the layers milestone 1 does not run. +// +// Each prints its own name and parks. That is the whole point: reaching one of these means the +// transport got further than expected and the next layer is now needed, which is information. A +// stub that returned 0 would turn that into a hang with no console output. +// --------------------------------------------------------------------------------------------- + +fn unimplemented(comptime name: []const u8) noreturn { + soc.rom.print("\r\n=== esp_hosted reached " ++ name ++ ", which this build does not implement.\r\n", .{}); + soc.rom.print("=== The transport got further than milestone 1. Implement it in src/net/.\r\n", .{}); + while (true) {} +} + +// rpc_start and serial_ll_rx_handler are no longer stubbed here: build.zig compiles ESP-Hosted's +// own RPC layer (host/drivers/rpc/**, plus protobuf-c and the generated descriptors), which defines +// both. The loud stub did its job first - it is what turned "the radio hangs" into a console line +// naming rpc_start as the next thing to build. + +// Bluetooth is not stubbed here. ESP-Hosted ships host/drivers/bt/hci_stub_drv.c, which is the +// vendor's own no-op hci_drv_init and a drop-everything hci_rx_handler for a host without BT, and +// build.zig compiles it. Defining them here as well is a duplicate-symbol link error - which is how +// this comment came to exist. BT is switched off in src/net/hosted/sdkconfig.h so that file does not +// pull NimBLE in. + +/// ESP-Hosted's console commands. There is no console component in this image. +export fn esp_hosted_cli_start() callconv(.c) c_int { + unimplemented("esp_hosted_cli_start"); +} + +export fn esp_hosted_cli_stop() callconv(.c) c_int { + unimplemented("esp_hosted_cli_stop"); +} + +// create_debugging_tasks is ESP-Hosted's own (host/utils/stats.c), compiled by build.zig. With the +// stats Kconfig options off it spawns nothing. + +/// IDF's internal Wi-Fi receive-callback registration, and it stays a loud stub deliberately: the +/// station frame path does not go through it, and it has never been reached. +/// +/// It was expected to be the seam. It is not. In the file set build.zig compiles, the only caller +/// is `transport_drv_remove_channel` (transport_drv.c:252), which unregisters on teardown - and +/// nothing in this project tears a channel down. `transport_drv_add_channel`, the registration +/// half, never mentions it (transport_drv.c:440-502): it stores the callback in `chan_arr[if_type]` +/// and `sdio_process_rx_task` calls it from there (sdio_drv.c:1394-1408). That channel callback is +/// the whole story, and src/net/link.zig is what registers it. +/// +/// This function exists because ESP-Hosted's C references the symbol and the link needs it. If it +/// is ever reached, the console line it prints is real information - something began tearing the +/// station channel down - and that is worth more than a silent zero. +export fn esp_wifi_internal_reg_rxcb(_: c_int, _: ?*anyopaque) callconv(.c) c_int { + unimplemented("esp_wifi_internal_reg_rxcb"); +} + +/// Host power-save. Not used: this board is mains-powered and the path adds a wakeup protocol +/// between the P4 and the C6 that nothing here needs. +export fn stop_host_power_save() callconv(.c) c_int { + unimplemented("stop_host_power_save"); +} + +export fn esp_hosted_woke_from_power_save() callconv(.c) bool { + // Answering this one honestly is better than parking: it is called on the normal boot path, + // and the truthful answer on a board that never sleeps is "no". + return false; +} + +export fn release_slave_reset_gpio_post_wakeup() callconv(.c) void { + // Same reasoning: only meaningful after a power-save wakeup, which cannot have happened. +} + +// --------------------------------------------------------------------------------------------- +// esp_netif, declined. +// +// ESP-Hosted's RPC layer asks IDF's network-interface layer whether an interface exists and whether +// it is up, before handing it a received frame. This project does not use esp_netif or lwIP - the +// whole point of src/net/ip.zig is to replace them - so there is no handle to give it and no +// interface it would recognise. +// +// Answering "no interface" is the truthful answer and it is safe, and this is now observed rather +// than hoped for: station frames reach this project through the transport's own channel callback, +// which src/net/link.zig registers with `transport_drv_add_channel` and which sdio_drv.c:1394-1408 +// dispatches to. That path does not consult esp_netif at all. These two stay as they are. +// --------------------------------------------------------------------------------------------- + +export fn esp_netif_get_handle_from_ifkey(_: ?[*:0]const u8) callconv(.c) ?*anyopaque { + return null; +} + +export fn esp_netif_is_netif_up(_: ?*anyopaque) callconv(.c) bool { + return false; +} + +/// mempool.c's pluggable backend. ESP-Hosted's own static pool is used, so the ops table is null; +/// mempool.c checks for null and falls back. +export fn os_mempool_get_ops() callconv(.c) ?*anyopaque { + return null; +} + +// There are no tests in this file, and that is a deliberate answer rather than an omission. +// +// Everything here ends in the ROM UART printer (`ets_printf`, a mask-ROM address) or in +// `vsnprintf` from src/net/libc.zig, so a standalone host build compiles but cannot link. What is +// worth checking is the level *ordering* - and that is checkable at compile time, on every build, +// which is strictly better than a test that only runs when someone asks: + +comptime { + // IDF's esp_log_level_t numbers levels so that a HIGHER value is MORE verbose. The filter in + // `esp_log` above is therefore `msg_level > level -> drop`. Inverting that inequality would + // silently discard exactly the transport diagnostics that bring-up depends on, and the code + // would look right. These assertions pin the ordering the filter assumes. + std.debug.assert(@intFromEnum(Level.none) < @intFromEnum(Level.err)); + std.debug.assert(@intFromEnum(Level.err) < @intFromEnum(Level.warn)); + std.debug.assert(@intFromEnum(Level.warn) < @intFromEnum(Level.info)); + std.debug.assert(@intFromEnum(Level.info) < @intFromEnum(Level.debug)); + std.debug.assert(@intFromEnum(Level.debug) < @intFromEnum(Level.verbose)); + + // The values must be IDF's own, not merely ordered: ESP-Hosted's C passes esp_log_level_t + // integers across the ABI, so a shifted enum would misclassify every line. + std.debug.assert(@intFromEnum(Level.err) == 1); + std.debug.assert(@intFromEnum(Level.verbose) == 5); +} diff --git a/src/net/hosted_os.zig b/src/net/hosted_os.zig new file mode 100644 index 0000000..d844177 --- /dev/null +++ b/src/net/hosted_os.zig @@ -0,0 +1,890 @@ +//! ESP-Hosted's OS objects - mutex, counting semaphore, fixed-capacity queue, thread, software +//! timer - expressed in `std.Io`, with FreeRTOS's exact observable behaviour. +//! +//! Nothing here reimplements a synchronisation primitive. `std.Io.Mutex`, `std.Io.Semaphore` and +//! `std.Io.TypeErasedQueue` do the blocking; this file supplies only the three things ESP-Hosted +//! needs that they do not have: +//! +//! 1. **The timeout dialect.** `_h_lock_mutex`, `_h_get_semaphore` and `_h_dequeue_item` all take +//! an `int`, where 0 means "do not block", a negative value means "block forever", and a +//! positive value means a bounded wait. The unit of that positive value is *not* the same in +//! all three - see `Wait`. +//! 2. **The return codes.** `RET_OK`/`RET_FAIL`/`RET_INVALID`/`RET_FAIL_TIMEOUT` from +//! `port_esp_hosted_host_os.h:86-91`, which the C caller branches on. +//! 3. **The initial state.** A FreeRTOS semaphore created by +//! `hosted_create_semaphore` (`port_esp_hosted_host_os.c:523-547`) is given *once* before it +//! is returned, so it starts with one permit, and callers rely on that: `sdio_drv.c:1504`, +//! `:1508` and `:1540` each take that permit back immediately after creating the semaphore. A +//! semaphore that started at zero would leave every count in the transport off by one. +//! +//! This file is deliberately free of hardware and of the C ABI, so it runs on the host under +//! `std.Io.Threaded` and the tests below are real tests. + +const std = @import("std"); +const assert = std.debug.assert; +const Io = std.Io; +const Allocator = std.mem.Allocator; + +/// `port_esp_hosted_host_os.h:86-91`. +pub const ret = struct { + pub const ok: c_int = 0; + pub const fail: c_int = -1; + pub const invalid: c_int = -2; + pub const fail_mem: c_int = -3; + pub const fail4: c_int = -4; + pub const fail_timeout: c_int = -5; +}; + +/// The clock everything here measures against. `.awake` is `std.Io`'s monotonic clock; on this +/// board it is `hal.systimer`'s fixed 16 MHz, which does not move when the CPU clock does. +pub const clock: Io.Clock = .awake; + +/// How often a bounded wait re-checks. +/// +/// `std.Io.Semaphore` and `std.Io.TypeErasedQueue` have no timed acquire, and neither does +/// `std.Io.Mutex`; only `futexWaitTimeout` does, and reaching for it would mean rebuilding those +/// three primitives instead of using them. So a *bounded* wait polls, and an unbounded one - which +/// is what every hot path in ESP-Hosted actually uses - blocks properly with no polling at all. +/// +/// The cost is bounded and small: a bounded wait is used in exactly one place in the tree, +/// `rpc_core.c:844`, the synchronous-RPC response wait, whose timeout is measured in seconds. One +/// millisecond of added latency on a request that is allowed to take five seconds is not worth a +/// hand-rolled futex semaphore. +pub const poll_interval_ms: u32 = 1; + +/// The three shapes an ESP-Hosted timeout argument can take. +pub const Wait = union(enum) { + /// `0` - try, do not block. + immediate, + /// Negative, i.e. `HOSTED_BLOCKING` (-1) or `HOSTED_BLOCK_MAX` (`portMAX_DELAY`, which reaches + /// an `int` parameter as -1). + forever, + /// A bounded wait, in milliseconds. + bounded_ms: u32, + + /// The dialect used by `_h_lock_mutex` and `_h_get_semaphore`: a positive value is + /// milliseconds (`port_esp_hosted_host_os.c:452`, `:573`). + pub fn fromMillis(timeout: c_int) Wait { + if (timeout == 0) return .immediate; + if (timeout < 0) return .forever; + return .{ .bounded_ms = @intCast(timeout) }; + } + + /// The dialect used by `_h_dequeue_item`: a positive value is *seconds*, because the + /// implementation converts it with `SEC_TO_MILLISEC` before `pdMS_TO_TICKS` + /// (`port_esp_hosted_host_os.c:336`). The asymmetry with `fromMillis` is not a mistake in this + /// file; it is a mistake in ESP-Hosted that this file has to reproduce. No caller in the tree + /// passes a positive value to a queue, so nothing depends on it today. + pub fn fromQueueTimeout(timeout: c_int) Wait { + if (timeout == 0) return .immediate; + if (timeout < 0) return .forever; + return .{ .bounded_ms = @as(u32, @intCast(timeout)) *| 1000 }; + } +}; + +/// Milliseconds on `clock`, which is what `_h_get_time_ms` returns. +/// +/// The narrowing to `u64` before the division is not cosmetic. `Io.Timestamp.nanoseconds` is `i96`, +/// and `@divFloor` on an `i96` compiles to a call to compiler_rt's `__divti3` - a 128-bit software +/// division, on every call, on a 90 MHz in-order core. Narrowing first turns that into +/// `__udivdi3`, a 64-bit one. Both were read out of the object's undefined-symbol list rather than +/// guessed; ReleaseSmall declines to strength-reduce even a constant 64-bit divisor, so the +/// libcall stays, but it is now half the width. The range given up is imaginary: 2^64 nanoseconds +/// is 584 years, and this clock starts at boot. +pub fn nowMs(io: Io) u64 { + const ns = clock.now(io).nanoseconds; + if (ns <= 0) return 0; + const ns64: u64 = @intCast(ns); + return ns64 / std.time.ns_per_ms; +} + +/// Sleep one poll interval. Reports cancelation so bounded waits abandon promptly rather than +/// spinning out the full timeout after the task has been asked to stop. +fn pollSleep(io: Io) error{Canceled}!void { + return io.sleep(.fromMilliseconds(poll_interval_ms), clock); +} + +// --------------------------------------------------------------------------------------- Mutex + +/// FreeRTOS gives ESP-Hosted a *recursive-capable* mutex handle but ESP-Hosted never recurses on +/// one: every use is a bracketed `SDIO_LOCK`/`SDIO_UNLOCK` or equivalent, and every one of the ten +/// call sites in the tree passes `HOSTED_BLOCK_MAX`. So a plain `std.Io.Mutex` is the whole +/// requirement. +pub const Mutex = struct { + inner: Io.Mutex = .init, + + pub fn lock(m: *Mutex, io: Io, w: Wait) c_int { + switch (w) { + .immediate => return if (m.inner.tryLock()) ret.ok else ret.fail, + .forever => { + m.inner.lockUncancelable(io); + return ret.ok; + }, + .bounded_ms => |ms| { + const deadline = nowMs(io) + ms; + while (true) { + if (m.inner.tryLock()) return ret.ok; + if (nowMs(io) >= deadline) return ret.fail; + pollSleep(io) catch return ret.fail; + } + }, + } + } + + pub fn unlock(m: *Mutex, io: Io) c_int { + m.inner.unlock(io); + return ret.ok; + } +}; + +// ----------------------------------------------------------------------------------- Semaphore + +/// A counting semaphore with FreeRTOS's cap and FreeRTOS's initial count. +/// +/// The blocking path is `std.Io.Semaphore` untouched. What is added around it: +/// +/// * a **maximum count**, because `xSemaphoreCreateCounting(maxCount, 0)` refuses a give past +/// `maxCount` and `std.Io.Semaphore` has no ceiling. `sdio_drv.c:1502` sizes +/// `sem_to_slave_queue` at `tx_queue_size * MAX_PRIORITY_QUEUES` precisely so that the +/// semaphore saturates when the queues do. +/// * a **non-blocking take**, which `std.Io.Semaphore` does not expose. It is the tail of +/// `Semaphore.wait` (`std/Io/Semaphore.zig:18-24`) with the `cond.wait` loop removed, using +/// the same public fields, so it takes and releases the same mutex in the same order. +/// * an **ISR-deferred post**; see `postFromIsr`. +pub const Semaphore = struct { + inner: Io.Semaphore, + max: u32, + /// Posts an interrupt handler could not deliver directly. Folded in by the next task-side + /// operation on this semaphore. + isr_posts: std.atomic.Value(u32) = .init(0), + + /// `maxCount` as ESP-Hosted passes it: `<= 1` means a binary semaphore. + /// + /// Starts with one permit, matching `port_esp_hosted_host_os.c:544` - see the file header. + pub fn init(max_count: u32) Semaphore { + return .{ .inner = .{ .permits = 1 }, .max = @max(max_count, 1) }; + } + + pub fn post(s: *Semaphore, io: Io) c_int { + // Fold in anything an interrupt deferred, so every task-side entry point closes that + // window and not just the waiting ones. Two instructions when nothing is pending. + s.drainIsrPosts(io); + return if (s.add(io, 1) == 1) ret.ok else ret.fail; + } + + /// `_h_post_semaphore_from_isr`, and the one entry in the whole table whose FreeRTOS meaning + /// does not survive the move to a cooperative scheduler intact. + /// + /// FreeRTOS has `xSemaphoreGiveFromISR`, which manipulates the semaphore inside a port-level + /// critical section and then asks for a context switch on return from the interrupt. Neither + /// half exists here. `std.Io.Semaphore.post` takes the semaphore's own `Io.Mutex`, and an + /// interrupt that blocked on a mutex held by the task it interrupted would deadlock the core - + /// there is no other task to run and no preemption to run it. + /// + /// What is safe on this runtime, confirmed with the runtime's author: `io.futexWake` runs + /// inside a critical section that clears `mstatus.MIE`, touches only the run queue, and never + /// takes a task-held lock. `Io.Mutex.tryLock` is a single compare-exchange. So: + /// + /// * if the mutex is free, the post happens inline and completely. On a single core with + /// interrupts already masked, no task can observe the intermediate state. + /// * if the mutex is held, the interrupted task is *running* and holds it - `Io.Condition` + /// releases the mutex before it blocks (`std/Io.zig:1689`), so nobody ever sleeps holding + /// it. The post is recorded in `isr_posts` and folded in by that task's next operation on + /// this semaphore, which is a few instructions away. + /// + /// The residual hole: if the interrupt lands in that few-instruction window *and* the only + /// other participant is already blocked in `wait`, the deferred post sits until someone else + /// touches the semaphore. `drainIsrPosts` exists so an application can close it from an + /// interrupt epilogue. On the SDIO transport this is moot: `_h_post_semaphore_from_isr` has + /// exactly two callers in the tree, `spi_drv.c:181` and `:190`, plus `spi_hd_drv.c:174`, and + /// none of them is compiled for SDIO. + /// + /// Returns `RET_OK` if the post was delivered or deferred, never fails: an interrupt has + /// nowhere to report a failure to. + pub fn postFromIsr(s: *Semaphore, io: Io) c_int { + if (s.inner.mutex.tryLock()) { + defer s.inner.mutex.unlock(io); + if (s.inner.permits < s.max) { + s.inner.permits += 1; + s.inner.cond.signal(io); + } + return ret.ok; + } + _ = s.isr_posts.fetchAdd(1, .release); + return ret.ok; + } + + /// Fold any interrupt-deferred posts into the semaphore. Safe and cheap to call from a task at + /// any time; a no-op when nothing is pending. + pub fn drainIsrPosts(s: *Semaphore, io: Io) void { + const pending = s.isr_posts.swap(0, .acquire); + if (pending != 0) _ = s.add(io, pending); + } + + /// Add `n` permits, saturating at `max`. Returns how many were actually added. + fn add(s: *Semaphore, io: Io, n: u32) u32 { + s.inner.mutex.lockUncancelable(io); + defer s.inner.mutex.unlock(io); + const room = s.max -| @as(u32, @intCast(s.inner.permits)); + const added = @min(room, n); + if (added == 0) return 0; + s.inner.permits += added; + // One signal per permit: `Io.Condition.signal` releases exactly one waiter. + for (0..added) |_| s.inner.cond.signal(io); + return added; + } + + /// `_h_get_semaphore`. Returns 0 on success and `RET_FAIL_TIMEOUT` otherwise, which is what + /// `port_esp_hosted_host_os.c:577-579` returns and what `rpc_core.c:844` tests. + pub fn wait(s: *Semaphore, io: Io, w: Wait) c_int { + switch (w) { + .immediate => return if (s.tryTake(io)) ret.ok else ret.fail_timeout, + .forever => { + s.drainIsrPosts(io); + // Cancelation is reported as RET_FAIL_TIMEOUT, which is the only failure code + // `hosted_get_semaphore` ever returns (port_esp_hosted_host_os.c:579) and + // therefore the only one callers test for. + s.inner.wait(io) catch return ret.fail_timeout; + return ret.ok; + }, + .bounded_ms => |ms| { + const deadline = nowMs(io) + ms; + while (true) { + if (s.tryTake(io)) return ret.ok; + if (nowMs(io) >= deadline) return ret.fail_timeout; + pollSleep(io) catch return ret.fail; + } + }, + } + } + + /// Take a permit if one is available. The body is `Semaphore.wait` + /// (`std/Io/Semaphore.zig:18-24`) minus its `cond.wait` loop. + pub fn tryTake(s: *Semaphore, io: Io) bool { + s.drainIsrPosts(io); + s.inner.mutex.lockUncancelable(io); + defer s.inner.mutex.unlock(io); + if (s.inner.permits == 0) return false; + s.inner.permits -= 1; + if (s.inner.permits > 0) s.inner.cond.signal(io); + return true; + } + + pub fn count(s: *Semaphore, io: Io) u32 { + s.inner.mutex.lockUncancelable(io); + defer s.inner.mutex.unlock(io); + return @intCast(s.inner.permits); + } +}; + +// --------------------------------------------------------------------------------------- Queue + +/// A fixed-capacity queue of runtime-sized items. +/// +/// `_h_create_queue(qnum_elem, qitem_size)` fixes the element size at *run* time, so +/// `std.Io.Queue(Elem)` - which needs the type at compile time - cannot be used, but +/// `std.Io.TypeErasedQueue` can: it is a byte ring with `min`-byte put and get, which is exactly a +/// queue of fixed-size records once every operation moves `item_size` bytes. +/// +/// That the ring only ever moves whole items is what makes the non-blocking forms exact. The +/// buffer is `count * item_size` bytes and every transfer is `item_size`, so the occupied length is +/// always a multiple of `item_size`; a `min = 0` put therefore either fits the whole item or moves +/// nothing at all, and can never leave half a record in the ring. +pub const Queue = struct { + inner: Io.TypeErasedQueue, + item_size: u32, + /// Owned; freed by `destroy`. + buffer: []u8, + + pub fn create(gpa: Allocator, count: u32, item_size: u32) ?*Queue { + assert(item_size > 0); + const q = gpa.create(Queue) catch return null; + const buf = gpa.alloc(u8, @as(usize, count) * item_size) catch { + gpa.destroy(q); + return null; + }; + q.* = .{ .inner = .init(buf), .item_size = item_size, .buffer = buf }; + return q; + } + + pub fn destroy(q: *Queue, io: Io, gpa: Allocator) void { + q.inner.close(io); + gpa.free(q.buffer); + gpa.destroy(q); + } + + /// `_h_queue_item`. `item` points at one `item_size` record, which is copied into the queue - + /// FreeRTOS's `xQueueSendToBack` copies too, which is why every caller passes `&handle` rather + /// than a heap pointer. + pub fn send(q: *Queue, io: Io, item: [*]const u8, w: Wait) c_int { + const n = q.item_size; + const slice = item[0..n]; + switch (w) { + .immediate => { + const put = q.inner.put(io, slice, 0) catch return ret.fail; + return if (put == n) ret.ok else ret.fail; + }, + .forever => { + // Uncancelable, deliberately. A cancelable blocking put can be interrupted + // *after* it has copied part of a record into the ring, and since the ring's + // occupied length is what makes the non-blocking forms exact, a half record + // desynchronises every subsequent transfer. FreeRTOS's portMAX_DELAY does not + // return early either. The cost is that a canceled task blocked here stays + // blocked - which it would anyway: `Future.cancel` signals only the *next* + // cancelation point, and ESP-Hosted's task bodies loop straight back into the + // queue. See `Thread.cancel`. + const put = q.inner.putUncancelable(io, slice, n) catch return ret.fail; + return if (put == n) ret.ok else ret.fail; + }, + .bounded_ms => |ms| { + const deadline = nowMs(io) + ms; + while (true) { + const put = q.inner.put(io, slice, 0) catch return ret.fail; + if (put == n) return ret.ok; + assert(put == 0); // a partial record would corrupt the ring + if (nowMs(io) >= deadline) return ret.fail; + pollSleep(io) catch return ret.fail; + } + }, + } + } + + /// `_h_dequeue_item`. Returns 0 on success, `RET_FAIL` on timeout - note the asymmetry with + /// `Semaphore.wait`, which returns `RET_FAIL_TIMEOUT`; `port_esp_hosted_host_os.c:342` really + /// does return the plain failure code here. + pub fn receive(q: *Queue, io: Io, out: [*]u8, w: Wait) c_int { + const n = q.item_size; + const slice = out[0..n]; + switch (w) { + .immediate => { + const got = q.inner.get(io, slice, 0) catch return ret.fail; + return if (got == n) ret.ok else ret.fail; + }, + .forever => { + // Uncancelable for the same reason as `send`. + const got = q.inner.getUncancelable(io, slice, n) catch return ret.fail; + return if (got == n) ret.ok else ret.fail; + }, + .bounded_ms => |ms| { + const deadline = nowMs(io) + ms; + while (true) { + const got = q.inner.get(io, slice, 0) catch return ret.fail; + if (got == n) return ret.ok; + assert(got == 0); + if (nowMs(io) >= deadline) return ret.fail; + pollSleep(io) catch return ret.fail; + } + }, + } + } + + /// `_h_queue_msg_waiting` = `uxQueueMessagesWaiting`, which counts *buffered* items only and + /// not producers blocked with an item in hand. + pub fn waiting(q: *Queue, io: Io) c_int { + q.inner.mutex.lockUncancelable(io); + defer q.inner.mutex.unlock(io); + return @intCast(q.inner.len / q.item_size); + } + + /// `_h_reset_queue` = `xQueueReset`: discard everything buffered. Blocked producers and + /// consumers are left alone, which is also what FreeRTOS does for waiting *receivers*; it + /// differs in that FreeRTOS re-evaluates blocked senders. No caller in the tree uses this. + pub fn reset(q: *Queue, io: Io) c_int { + q.inner.mutex.lockUncancelable(io); + defer q.inner.mutex.unlock(io); + q.inner.start = 0; + q.inner.len = 0; + return ret.ok; + } +}; + +// -------------------------------------------------------------------------------------- Thread + +/// ESP-Hosted's task entry point: `void (*)(void const *)`, called once and never expected to +/// return (`port_esp_hosted_host_os.c:163`, and every body in the tree is a `while (1)` loop). +pub const StartRoutine = *const fn (?*const anyopaque) callconv(.c) void; + +pub const Thread = struct { + future: Io.Future(void), + name: [*:0]const u8, + + fn trampoline(start: StartRoutine, arg: ?*const anyopaque) void { + start(arg); + } + + /// `io.concurrent`, not `io.async`, and the difference is the whole point of the entry. + /// + /// `xTaskCreate` returns a task that exists and will run whatever its creator does next. + /// `io.async` promises less: the implementation is allowed to run the body inline before + /// returning, which for an ESP-Hosted task body - an unconditional `while (1)` - would never + /// return and would deadlock initialisation on the spot. `io.concurrent` forbids exactly that + /// (`std/Io.zig:2358-2364`) and reports `error.ConcurrencyUnavailable` when no unit of + /// concurrency is free. + /// + /// Turning that error into NULL is right: `_h_thread_create` is documented to return NULL on + /// failure and its callers check (`rpc_core.c:582`, `sdio_drv.c:1543`). A task pool one slot + /// too small then produces a legible "thread creation failed" instead of a hang. + pub fn create(io: Io, gpa: Allocator, name: [*:0]const u8, start: StartRoutine, arg: ?*const anyopaque) ?*Thread { + const t = gpa.create(Thread) catch return null; + t.* = .{ + .future = io.concurrent(trampoline, .{ start, arg }) catch { + gpa.destroy(t); + return null; + }, + .name = name, + }; + return t; + } + + /// `_h_thread_cancel` maps to `Future.cancel`, and this is the second place FreeRTOS's model + /// does not fit. + /// + /// `vTaskDelete` destroys a task from outside, wherever it happens to be. `std.Io`'s cancel is + /// cooperative: it asks, then *waits for the task body to return*. ESP-Hosted's task bodies + /// never return - `sdio_read_task`, `rpc_rx_thread` and the rest are unconditional loops - so + /// this call completes only if the body happens to exit, and otherwise blocks. + /// + /// That is survivable because of where it is called from: `cancel_rpc_threads` + /// (`rpc_core.c`) and the transport teardown paths, both of which run only when the host is + /// about to restart the slave. It is not survivable as a routine operation, and if a teardown + /// path becomes routine the fix is a `killTask` on the runtime that reclaims the slot without + /// unwinding, not a change here: there is no way to unwind a C frame from Zig. + pub fn cancel(t: *Thread, io: Io, gpa: Allocator) c_int { + t.future.cancel(io); + gpa.destroy(t); + return ret.ok; + } +}; + +// ------------------------------------------------------------------------------- software timers + +pub const TimerHandler = *const fn (?*anyopaque) callconv(.c) void; + +pub const TimerKind = enum(c_int) { + /// `H_TIMER_TYPE_ONESHOT`, port_esp_hosted_host_os.h:39. + oneshot = 0, + /// `H_TIMER_TYPE_PERIODIC`. + periodic = 1, +}; + +pub const Timer = struct { + handler: TimerHandler = undefined, + arg: ?*anyopaque = null, + /// Absolute deadline on `clock`, in milliseconds. + deadline_ms: u64 = 0, + /// 0 for a one-shot. + period_ms: u32 = 0, + in_use: bool = false, +}; + +/// One task servicing every software timer, rather than one task per timer. +/// +/// ESP-IDF backs `_h_timer_start` with `esp_timer`, which has its own dedicated task. Doing the +/// obvious thing here - `io.async` per timer - would cost one whole task slot and one whole static +/// stack per timer, and ESP-Hosted starts up to three concurrently: the slave-unresponsive timer +/// (`transport_drv.c:188`), the per-request asynchronous RPC timeout (`rpc_core.c:215`), and the +/// power-save timer. At the stack sizes this runtime needs that is 15 KB to run three sleeps. +/// +/// So: one task, an array of slots, and a futex word that a `start` or `stop` bumps to make the +/// service task recompute its next deadline. Static footprint is `@sizeOf(Timer)` (24 bytes on +/// rv32) per slot plus one task stack. +/// +/// Handlers run on the service task, not in an interrupt, so they may block. `init_timeout_cb` +/// (`transport_drv.c`) calls `_h_restart_host`, which never returns, and that is fine here. +pub fn TimerService(comptime slot_count: usize) type { + return struct { + const Self = @This(); + + slots: [slot_count]Timer = @splat(.{}), + /// Bumped whenever a slot is armed or disarmed; the service task waits on it. + epoch: std.atomic.Value(u32) = .init(0), + /// Guards `slots`. A plain `Io.Mutex`: every critical section here is a few dozen + /// instructions and never blocks. + mutex: Io.Mutex = .init, + task: ?*Thread = null, + stopping: bool = false, + + pub fn start(self: *Self, io: Io, gpa: Allocator) bool { + if (self.task != null) return true; + const t = gpa.create(Thread) catch return false; + // Concurrent for the same reason as `Thread.create`: the service loop never returns, + // so an implementation permitted to run it inline would never return from `start`. + t.* = .{ + .future = io.concurrent(service, .{ self, io }) catch { + gpa.destroy(t); + return false; + }, + .name = "hosted_timers", + }; + self.task = t; + return true; + } + + /// Arm a slot. Returns its index, or null when every slot is in use. + pub fn arm(self: *Self, io: Io, ms: u32, kind: TimerKind, handler: TimerHandler, arg: ?*anyopaque) ?usize { + self.mutex.lockUncancelable(io); + const idx = blk: { + for (&self.slots, 0..) |*s, i| if (!s.in_use) break :blk i; + self.mutex.unlock(io); + return null; + }; + self.slots[idx] = .{ + .handler = handler, + .arg = arg, + .deadline_ms = nowMs(io) + ms, + .period_ms = if (kind == .periodic) ms else 0, + .in_use = true, + }; + self.mutex.unlock(io); + self.kick(io); + return idx; + } + + pub fn disarm(self: *Self, io: Io, idx: usize) c_int { + if (idx >= slot_count) return ret.invalid; + self.mutex.lockUncancelable(io); + const was = self.slots[idx].in_use; + self.slots[idx].in_use = false; + self.mutex.unlock(io); + self.kick(io); + return if (was) ret.ok else ret.fail; + } + + fn kick(self: *Self, io: Io) void { + _ = self.epoch.fetchAdd(1, .release); + io.futexWake(u32, &self.epoch.raw, 1); + } + + fn service(self: *Self, io: Io) void { + while (!self.stopping) { + const seen = self.epoch.load(.acquire); + const now = nowMs(io); + + // Fire everything due, collecting the handlers first so none of them runs while + // the slot table is locked: a handler may arm or disarm a timer. + var due: [slot_count]struct { h: TimerHandler, a: ?*anyopaque } = undefined; + var due_len: usize = 0; + var next_deadline: ?u64 = null; + + self.mutex.lockUncancelable(io); + for (&self.slots) |*s| { + if (!s.in_use) continue; + if (s.deadline_ms <= now) { + due[due_len] = .{ .h = s.handler, .a = s.arg }; + due_len += 1; + if (s.period_ms == 0) { + s.in_use = false; + } else { + s.deadline_ms = now + s.period_ms; + } + } + if (s.in_use) { + if (next_deadline == null or s.deadline_ms < next_deadline.?) + next_deadline = s.deadline_ms; + } + } + self.mutex.unlock(io); + + for (due[0..due_len]) |d| d.h(d.a); + if (due_len != 0) continue; + + if (next_deadline) |dl| { + const remaining = dl -| nowMs(io); + io.futexWaitTimeout(u32, &self.epoch.raw, seen, .{ + .duration = .{ .clock = clock, .raw = .fromMilliseconds(@intCast(remaining)) }, + }) catch return; + } else { + io.futexWait(u32, &self.epoch.raw, seen) catch return; + } + } + } + + pub fn stop(self: *Self, io: Io, gpa: Allocator) void { + const t = self.task orelse return; + self.stopping = true; + self.kick(io); + t.future.cancel(io); + gpa.destroy(t); + self.task = null; + } + }; +} + +// ---------------------------------------------------------------------------------------- tests + +const testing = std.testing; + +fn hostIo() struct { threaded: *Io.Threaded, io: Io } { + const t = testing.allocator.create(Io.Threaded) catch unreachable; + t.* = .init(testing.allocator, .{}); + return .{ .threaded = t, .io = t.io() }; +} + +test "Semaphore starts with one permit, as FreeRTOS's create+give does" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + + // sdio_drv.c:1502-1504 creates a counting semaphore and immediately takes the permit that + // hosted_create_semaphore left behind. If the count started at zero this take would fail and + // every subsequent count would be one too high. + var s = Semaphore.init(60); + try testing.expectEqual(@as(u32, 1), s.count(io)); + try testing.expectEqual(ret.ok, s.wait(io, .immediate)); + try testing.expectEqual(@as(u32, 0), s.count(io)); + + // Empty: a non-blocking take reports RET_FAIL_TIMEOUT, which is the code rpc_core.c:844 tests. + try testing.expectEqual(ret.fail_timeout, s.wait(io, .immediate)); +} + +test "Semaphore counts, saturates at max, and times out" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + + var s = Semaphore.init(3); + // Starts at 1; two more posts reach the cap. + try testing.expectEqual(ret.ok, s.post(io)); + try testing.expectEqual(ret.ok, s.post(io)); + try testing.expectEqual(@as(u32, 3), s.count(io)); + // FreeRTOS's xSemaphoreGive returns pdFALSE past maxCount, and so does this. + try testing.expectEqual(ret.fail, s.post(io)); + try testing.expectEqual(@as(u32, 3), s.count(io)); + + for (0..3) |_| try testing.expectEqual(ret.ok, s.wait(io, .immediate)); + + // A bounded wait on an empty semaphore returns RET_FAIL_TIMEOUT, and takes at least as long as + // it was asked to. + const before = nowMs(io); + try testing.expectEqual(ret.fail_timeout, s.wait(io, .{ .bounded_ms = 25 })); + try testing.expect(nowMs(io) - before >= 25); +} + +test "Semaphore: a blocked waiter is released by a post from another task" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + + var s = Semaphore.init(4); + try testing.expectEqual(ret.ok, s.wait(io, .immediate)); // drain the initial permit + + const Worker = struct { + fn run(sem: *Semaphore, i: Io) c_int { + return sem.wait(i, .forever); + } + }; + var f = io.async(Worker.run, .{ &s, io }); + // Give the waiter time to actually block, then release it. + try io.sleep(.fromMilliseconds(20), clock); + try testing.expectEqual(ret.ok, s.post(io)); + try testing.expectEqual(ret.ok, f.await(io)); + try testing.expectEqual(@as(u32, 0), s.count(io)); +} + +test "Semaphore: an interrupt-deferred post is folded in by the next task-side operation" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + + var s = Semaphore.init(4); + try testing.expectEqual(ret.ok, s.wait(io, .immediate)); + + // Simulate the contended case: hold the semaphore's mutex, so postFromIsr cannot deliver + // inline and must defer. This is the exact window described on `postFromIsr`. + s.inner.mutex.lockUncancelable(io); + try testing.expectEqual(ret.ok, s.postFromIsr(io)); + try testing.expectEqual(@as(u32, 1), s.isr_posts.load(.acquire)); + s.inner.mutex.unlock(io); + + // The next task-side touch delivers it. + try testing.expectEqual(ret.ok, s.wait(io, .immediate)); + try testing.expectEqual(@as(u32, 0), s.isr_posts.load(.acquire)); + + // Uncontended, it lands directly. + try testing.expectEqual(ret.ok, s.postFromIsr(io)); + try testing.expectEqual(@as(u32, 0), s.isr_posts.load(.acquire)); + try testing.expectEqual(@as(u32, 1), s.count(io)); +} + +test "Queue: fixed-capacity records, non-blocking edges, and message count" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + const gpa = testing.allocator; + + // 24 bytes is sizeof(interface_buffer_handle_t) on rv32, which is what every transport queue + // in ESP-Hosted carries. + const item_size = 24; + const q = Queue.create(gpa, 4, item_size).?; + defer q.destroy(io, gpa); + + try testing.expectEqual(@as(c_int, 0), q.waiting(io)); + // Empty, non-blocking: RET_FAIL, and note it is RET_FAIL and not RET_FAIL_TIMEOUT. + var out: [item_size]u8 = undefined; + try testing.expectEqual(ret.fail, q.receive(io, &out, .immediate)); + + var item: [item_size]u8 = undefined; + for (0..4) |i| { + @memset(&item, @intCast(i)); + try testing.expectEqual(ret.ok, q.send(io, &item, .immediate)); + try testing.expectEqual(@as(c_int, @intCast(i + 1)), q.waiting(io)); + } + // Full: a non-blocking send fails and leaves no partial record behind. + @memset(&item, 0xFF); + try testing.expectEqual(ret.fail, q.send(io, &item, .immediate)); + try testing.expectEqual(@as(c_int, 4), q.waiting(io)); + + // FIFO order, whole records. + for (0..4) |i| { + try testing.expectEqual(ret.ok, q.receive(io, &out, .forever)); + try testing.expect(std.mem.allEqual(u8, &out, @intCast(i))); + } + try testing.expectEqual(@as(c_int, 0), q.waiting(io)); + + // A bounded receive on an empty queue waits and then fails. + const before = nowMs(io); + try testing.expectEqual(ret.fail, q.receive(io, &out, .{ .bounded_ms = 25 })); + try testing.expect(nowMs(io) - before >= 25); +} + +test "Queue: blocking receive is woken by a producer, and reset discards" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + const gpa = testing.allocator; + + const q = Queue.create(gpa, 2, 4).?; + defer q.destroy(io, gpa); + + const Consumer = struct { + fn run(queue: *Queue, i: Io) u32 { + var buf: [4]u8 = undefined; + if (queue.receive(i, &buf, .forever) != ret.ok) return 0xDEAD; + return std.mem.readInt(u32, &buf, .little); + } + }; + var f = io.async(Consumer.run, .{ q, io }); + try io.sleep(.fromMilliseconds(20), clock); + + var word: [4]u8 = undefined; + std.mem.writeInt(u32, &word, 0xC0FFEE, .little); + try testing.expectEqual(ret.ok, q.send(io, &word, .forever)); + try testing.expectEqual(@as(u32, 0xC0FFEE), f.await(io)); + + // reset drops buffered records. + try testing.expectEqual(ret.ok, q.send(io, &word, .immediate)); + try testing.expectEqual(ret.ok, q.send(io, &word, .immediate)); + try testing.expectEqual(@as(c_int, 2), q.waiting(io)); + _ = q.reset(io); + try testing.expectEqual(@as(c_int, 0), q.waiting(io)); +} + +test "Mutex: the three timeout dialects" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + + var m: Mutex = .{}; + try testing.expectEqual(ret.ok, m.lock(io, .forever)); + // Held: a non-blocking lock fails rather than deadlocking. + try testing.expectEqual(ret.fail, m.lock(io, .immediate)); + const before = nowMs(io); + try testing.expectEqual(ret.fail, m.lock(io, .{ .bounded_ms = 25 })); + try testing.expect(nowMs(io) - before >= 25); + try testing.expectEqual(ret.ok, m.unlock(io)); + try testing.expectEqual(ret.ok, m.lock(io, .immediate)); + try testing.expectEqual(ret.ok, m.unlock(io)); +} + +test "Wait: the two timeout dialects ESP-Hosted uses" { + // _h_lock_mutex and _h_get_semaphore: positive means milliseconds. + try testing.expectEqual(Wait.immediate, Wait.fromMillis(0)); + try testing.expectEqual(Wait.forever, Wait.fromMillis(-1)); + // HOSTED_BLOCK_MAX is portMAX_DELAY, 0xFFFFFFFF, which reaches an `int` parameter as -1. + try testing.expectEqual(Wait.forever, Wait.fromMillis(@bitCast(@as(u32, 0xFFFF_FFFF)))); + try testing.expectEqual(Wait{ .bounded_ms = 5000 }, Wait.fromMillis(5000)); + + // _h_dequeue_item: positive means seconds. port_esp_hosted_host_os.c:336. + try testing.expectEqual(Wait{ .bounded_ms = 5000 }, Wait.fromQueueTimeout(5)); + try testing.expectEqual(Wait.forever, Wait.fromQueueTimeout(-1)); +} + +test "TimerService: one-shot fires once, periodic repeats, stop cancels" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + const gpa = testing.allocator; + + const Counter = struct { + var oneshot: u32 = 0; + var periodic: u32 = 0; + fn bumpOneshot(_: ?*anyopaque) callconv(.c) void { + oneshot += 1; + } + fn bumpPeriodic(_: ?*anyopaque) callconv(.c) void { + periodic += 1; + } + }; + Counter.oneshot = 0; + Counter.periodic = 0; + + var svc: TimerService(4) = .{}; + try testing.expect(svc.start(io, gpa)); + defer svc.stop(io, gpa); + + _ = svc.arm(io, 10, .oneshot, Counter.bumpOneshot, null).?; + const p = svc.arm(io, 10, .periodic, Counter.bumpPeriodic, null).?; + + try io.sleep(.fromMilliseconds(120), clock); + try testing.expectEqual(@as(u32, 1), Counter.oneshot); + try testing.expect(Counter.periodic >= 3); + + // Disarming stops it; the count must not move afterwards. + try testing.expectEqual(ret.ok, svc.disarm(io, p)); + const frozen = Counter.periodic; + try io.sleep(.fromMilliseconds(60), clock); + try testing.expectEqual(frozen, Counter.periodic); + // Disarming an already-disarmed slot reports failure, as esp_timer_stop does. + try testing.expectEqual(ret.fail, svc.disarm(io, p)); +} + +test "TimerService: slot exhaustion is reported, not fatal" { + var h = hostIo(); + defer { + h.threaded.deinit(); + testing.allocator.destroy(h.threaded); + } + const io = h.io; + + const Nop = struct { + fn f(_: ?*anyopaque) callconv(.c) void {} + }; + var svc: TimerService(2) = .{}; + _ = svc.arm(io, 10_000, .oneshot, Nop.f, null).?; + _ = svc.arm(io, 10_000, .oneshot, Nop.f, null).?; + try testing.expectEqual(@as(?usize, null), svc.arm(io, 10_000, .oneshot, Nop.f, null)); +} diff --git a/src/net/ip.zig b/src/net/ip.zig new file mode 100644 index 0000000..2cc4301 --- /dev/null +++ b/src/net/ip.zig @@ -0,0 +1,2903 @@ +//! A minimal IPv4 stack: Ethernet, ARP, IPv4, ICMP echo, UDP, a DHCP client, one TCP client and +//! HTTP GET. This is what replaces lwIP. +//! +//! Two functions drive everything and nothing else touches the outside world: +//! +//! stack.onFrame(frame) a received Ethernet frame, headers and all +//! stack.tick(now_ms) time passing, in milliseconds, from anywhere the caller likes +//! +//! and one callback carries frames out (`send`, supplied to `init`). There is no `std.Io`, no +//! allocator, no clock read and no hidden thread. That is not minimalism for its own sake: it is +//! what makes the whole stack testable on the host, where a "network" is a test function that hands +//! `onFrame` bytes it wrote by hand and reads back whatever `send` was given. Every protocol +//! behaviour in this file is exercised that way in `ip_test.zig`, including retransmission - which +//! on a real timer would be a flaky test and here is two calls to `tick`. +//! +//! **Everything is statically sized.** `Stack` is one struct with fixed buffers inside it; there is +//! no allocator, not even a `FixedBufferAllocator`, because nothing here has a lifetime that an +//! arena would model better than a field does. `@sizeOf(Stack)` is asserted at compile time below +//! (`footprint`) so the number cannot drift silently against the ~128 KB of L2MEM the image has. +//! +//! Wire formats are matched field by field against the lwIP this replaces, and every one is cited: +//! ESP-IDF v6.0.2 carries lwIP at `components/lwip/lwip/src/`, and the packed structs in +//! `include/lwip/prot/*.h` are the reference for offsets, and its `.c` files for behaviour. Where +//! this stack deliberately differs from lwIP, the comment says so and why. +//! +//! ## What this is not +//! +//! * No IPv6, no TCP listen/accept, no IP fragmentation or reassembly, no TLS. Out of scope. +//! * No congestion control. TCP sends at most one unacknowledged segment at a time (see `Tcp`), +//! which is a fixed window of one and therefore needs no congestion window, no slow start and +//! no fast recovery. It is also slow. For an HTTP GET of a few kilobytes over Wi-Fi that is the +//! right trade; for bulk transfer it is not, and nothing here pretends otherwise. +//! * No VLAN tags, no 802.1Q. A tagged frame is dropped as an unknown ethertype. +//! * No transfer coding but `identity` and `chunked`. Anything else - `gzip`, `deflate`, a +//! stack of them - is rejected with `error.UnsupportedTransferEncoding` rather than handed +//! back with its framing bytes still in it. +//! * DNS resolves A records only, one query at a time, over the UDP already here, with no +//! cache. `resolve` follows `httpGet`'s protocol exactly: start, `error.WouldBlock`, the +//! caller drives `tick`/`onFrame`, call again with the same name. + +const std = @import("std"); +const assert = std.debug.assert; + +// =============================================================================== sizing +// +// The whole static footprint, in one place. Every buffer in `Stack` is one of these. + +/// Ethernet MTU: the largest IP datagram that fits in one frame. +pub const mtu: usize = 1500; +/// Ethernet header: 6 destination + 6 source + 2 ethertype. lwIP `prot/ethernet.h:89` +/// (`SIZEOF_ETH_HDR`, with its optional `ETH_PAD_SIZE` at zero). +pub const eth_hlen: usize = 14; +/// The largest frame this stack will build or accept, excluding the FCS the MAC appends. +pub const frame_max: usize = eth_hlen + mtu; + +/// ARP cache entries. Four is enough for the gateway, one peer, and two strangers, which is the +/// whole population a single-connection HTTP client on a home /24 ever needs to address. +pub const arp_cache_len: usize = 4; + +/// Bytes of HTTP response head (status line plus headers) that may be buffered while waiting for +/// the blank line. Exceeding this fails the request rather than truncating silently. +/// +/// 2048, raised from 1024 against a measurement rather than a guess. A real response from the site +/// this stack was pointed at - Cloudflare in front of GitHub Pages - carries **1043 bytes** of head: +/// 26 header lines, of which `Report-To` alone is 254 bytes and `Nel`, `X-Fastly-Request-ID`, +/// `X-GitHub-Request-Id` and `alt-svc` are another 200 between them. At 1024 the request failed with +/// `HttpHeadersTooLong` after the body had already been negotiated, 19 bytes short. +/// +/// Modern CDN responses simply have large heads, and 1 KB is not a realistic ceiling for one. 2 KB +/// leaves about a kilobyte of margin over the measured case; the failure remains a named error +/// rather than truncation, so a head that exceeds even this is still diagnosable rather than silently +/// wrong. +pub const http_head_max: usize = 2048; + +/// Bytes of chunked *framing* - one chunk's extension parameters, or the whole trailer section - +/// tolerated before the response is failed. Framing is skipped rather than stored, so this bounds +/// work and not memory: without it a peer that streams `;a=b` forever, or trailer lines forever, +/// is a request that never ends and never errors. 512 is generous; a real trailer section is one +/// or two short lines. +pub const http_framing_max: usize = 512; + +/// The longest host name `resolve` will encode into a DNS question, in dotted text. RFC 1035 2.3.4 +/// allows 255; this stack holds the encoded question in `Stack` for the duration of the query, and +/// 64 covers every name a device that fetches one URL will ever ask for. A longer one is +/// `error.NameTooLong`, never a silently truncated question. +pub const dns_name_max: usize = 64; + +/// The encoded question that `dns_name_max` produces. Encoding turns `a.b` into `1a1b0`: one +/// length byte per label plus the root label, which for a name with no trailing dot is exactly +/// two bytes more than the text. RFC 1035 4.1.2. +pub const dns_qname_max: usize = dns_name_max + 2; + +/// Bytes of outbound TCP payload held for retransmission. This is sized for one HTTP request line +/// plus headers; there is no streaming send, so it is also the hard limit on request size. +pub const tcp_tx_max: usize = 512; + +/// The receive window this stack advertises, in bytes, when it has that much room to consume into. +/// One MSS: a peer that fills the window gets a segment acknowledged before it may send another. +pub const tcp_window: u16 = 1460; + +/// TCP MSS offered in the SYN. 1500 - 20 (IP) - 20 (TCP). +pub const tcp_mss: u16 = 1460; + +/// RFC 1122 4.2.2.6: a peer that sends no MSS option is assumed to accept 536. +pub const tcp_default_mss: u16 = 536; + +/// Initial retransmission timeout. RFC 6298 2.1 specifies 1 s for a connection with no RTT sample, +/// and this stack never takes an RTT sample - see `Tcp.rto_ms`. +pub const tcp_rto_initial_ms: u32 = 1000; +/// Retransmission timeout ceiling. RFC 6298 5.7 allows any value at or above 60 s; 16 s is chosen +/// against a device whose whole reason to exist is one short request. +pub const tcp_rto_max_ms: u32 = 16_000; +/// Retransmissions of the same segment before the connection is abandoned with `error.TimedOut`. +/// With the backoff above that is 1+2+4+8+16+16 = 47 s of trying. +pub const tcp_max_retries: u8 = 6; +/// TIME_WAIT duration. RFC 793 says 2*MSL, conventionally 240 s. Two seconds is what this uses: +/// holding a connection block for four minutes on a part with 128 KB of RAM to protect a +/// port number that this stack increments on every connect is the wrong trade. The risk it drops is +/// a late duplicate segment from the *previous* incarnation of the same 4-tuple being accepted into +/// a new one, and incrementing the local port already makes a repeat 4-tuple require 16,384 +/// connections first. +pub const tcp_time_wait_ms: u32 = 2000; +/// How long a half-closed connection waits for the peer's FIN before the block is released. RFC +/// 793 has no such timer and a connection may legitimately sit in FIN-WAIT-2 forever; Linux uses +/// 60 s for the same reason this uses 10 s - a peer that has our FIN and never answers is a peer +/// that is gone, and the one connection block here is not worth holding for it. +pub const tcp_fin_wait2_ms: u32 = 10_000; + +/// DHCP retransmission backoff, in milliseconds, indexed by attempt. RFC 2131 4.1 asks for +/// randomised exponential backoff starting at 4 s; this starts at 2 s because the first DHCP +/// exchange is on the critical path of every boot, and does not randomise because there is one +/// client on this board and the collision RFC 2131 is avoiding is between many. +const dhcp_backoff_ms = [_]u32{ 2_000, 4_000, 8_000, 16_000, 32_000, 64_000 }; + +/// Minimum length of the BOOTP/DHCP message this stack transmits, in UDP payload bytes. RFC 951 +/// fixed BOOTP messages at 300 bytes and relay agents in the field still expect at least that +/// much; lwIP pads the same way through its fixed-size `struct dhcp_msg` +/// (`prot/dhcp.h:63-91`: 236 + 4 cookie + `DHCP_OPTIONS_LEN` 68 = 308). +const dhcp_min_msg_len: usize = 300; + +/// DNS retransmission backoff, in milliseconds, indexed by attempt. RFC 1035 4.2.1 leaves the +/// timer to the implementation; this is BIND's classic 1 s doubling, and the array length is the +/// try count, so the whole exchange is bounded at 1+2+4 = 7 s and then `error.TimedOut`. The +/// transaction id is *not* redrawn between tries: a slow first answer must still be accepted. +const dns_backoff_ms = [_]u32{ 1_000, 2_000, 4_000 }; + +/// Compression pointers followed while skipping one name (RFC 1035 4.1.4). This is the bound that +/// makes a hostile message terminate: see `dnsSkipName`, where the argument is written out. +const dns_max_jumps: u8 = 16; + +// =============================================================== addresses and enumerations + +pub const Mac = [6]u8; +pub const Ip4 = [4]u8; + +pub const mac_broadcast: Mac = @splat(0xff); +pub const ip_any: Ip4 = @splat(0x00); +pub const ip_broadcast: Ip4 = @splat(0xff); + +/// Ethernet type field values. lwIP `prot/ieee.h:52-85` (`enum lwip_ieee_eth_type`). +pub const EtherType = enum(u16) { + ip4 = 0x0800, + arp = 0x0806, + vlan = 0x8100, + ip6 = 0x86dd, + _, +}; + +/// IP header protocol numbers. lwIP `prot/ip.h:46-50`. +pub const Protocol = enum(u8) { + icmp = 1, + tcp = 6, + udp = 17, + _, +}; + +// =============================================================================== checksum +// +// One implementation for IPv4, ICMP, UDP and TCP. The last two prepend a pseudo-header, which is +// the only difference between them: the arithmetic is identical, so it is written once. + +/// The Internet checksum of RFC 1071: the one's complement of the one's complement sum of the +/// data taken as 16-bit big-endian words, with a zero byte appended if the length is odd. +/// +/// Incremental, because TCP and UDP checksum a pseudo-header, a header and a payload that are +/// three separate buffers and never adjacent in memory. Feeding them in sequence must give the +/// same answer as checksumming the concatenation, which is why `half` exists: a chunk of odd +/// length leaves the high byte of a word owed, and the next chunk's first byte completes it. +/// Getting that wrong is invisible until a payload happens to have odd length, which for HTTP is +/// most of the time. +pub const Checksum = struct { + /// Accumulated 16-bit words. Deferring the end-around carry is safe for any length this + /// stack can produce: 32 bits absorbs 65,536 words, and the largest thing checksummed here is + /// 1,500 bytes. + sum: u32 = 0, + /// High byte of a 16-bit word whose low byte has not arrived yet. + half: ?u8 = null, + + pub fn update(self: *Checksum, bytes: []const u8) void { + var b = bytes; + if (self.half) |hi| { + if (b.len == 0) return; + self.sum += (@as(u32, hi) << 8) | b[0]; + self.half = null; + b = b[1..]; + } + var i: usize = 0; + while (i + 1 < b.len) : (i += 2) self.sum += std.mem.readInt(u16, b[i..][0..2], .big); + if (i < b.len) self.half = b[i]; + } + + /// Feed a big-endian 16-bit value, for the pseudo-header fields that are not in any buffer. + pub fn update16(self: *Checksum, v: u16) void { + var tmp: [2]u8 = undefined; + std.mem.writeInt(u16, &tmp, v, .big); + self.update(&tmp); + } + + /// The checksum as it goes on the wire. RFC 1071 1: an odd-length buffer is padded with a + /// zero byte, which the fold below does implicitly by shifting the owed byte up. + pub fn final(self: Checksum) u16 { + var s = self.sum; + if (self.half) |hi| s += @as(u32, hi) << 8; + while (s >> 16 != 0) s = (s & 0xffff) + (s >> 16); + return ~@as(u16, @truncate(s)); + } +}; + +/// The Internet checksum of one contiguous buffer. +pub fn checksum(bytes: []const u8) u16 { + var c: Checksum = .{}; + c.update(bytes); + return c.final(); +} + +/// The TCP/UDP pseudo-header of RFC 793 3.1: source address, destination address, a zero byte, the +/// protocol number and the transport length. Not transmitted; only checksummed. +fn pseudoHeader(c: *Checksum, src: Ip4, dst: Ip4, proto: Protocol, len: u16) void { + c.update(&src); + c.update(&dst); + c.update16(@intFromEnum(proto)); // the zero byte and the protocol byte, as one word + c.update16(len); +} + +/// A checksum for a UDP or TCP segment: pseudo-header, then the segment with its own checksum +/// field already zeroed. +fn transportChecksum(src: Ip4, dst: Ip4, proto: Protocol, segment: []const u8) u16 { + var c: Checksum = .{}; + pseudoHeader(&c, src, dst, proto, @intCast(segment.len)); + c.update(segment); + return c.final(); +} + +/// Verify a received transport checksum. A UDP datagram may carry zero, meaning "not computed" +/// (RFC 768); TCP may not. +fn transportChecksumOk(src: Ip4, dst: Ip4, proto: Protocol, segment: []const u8, field: u16) bool { + if (proto == .udp and field == 0) return true; + // Summing a segment that already contains its own checksum yields 0 (or, equivalently, the + // sum before complementing is 0xffff). RFC 1071 1. + return transportChecksum(src, dst, proto, segment) == 0; +} + +/// A transmitted UDP checksum of zero would be read as "not computed", so RFC 768 requires it be +/// sent as the equivalent 0xffff instead. TCP has no such rule and no such ambiguity. +pub fn udpChecksumOnWire(c: u16) u16 { + return if (c == 0) 0xffff else c; +} + +// ============================================================== unaligned big-endian access +// +// `std.mem.readInt`/`writeInt` with an explicit endianness at every single field. Never a shift and +// an or: a network header written by hand is where byte order goes wrong, and it goes wrong +// silently, on one field, in a way that looks like a hardware problem. + +inline fn rd16(b: []const u8, off: usize) u16 { + return std.mem.readInt(u16, b[off..][0..2], .big); +} +inline fn rd32(b: []const u8, off: usize) u32 { + return std.mem.readInt(u32, b[off..][0..4], .big); +} +inline fn wr16(b: []u8, off: usize, v: u16) void { + std.mem.writeInt(u16, b[off..][0..2], v, .big); +} +inline fn wr32(b: []u8, off: usize, v: u32) void { + std.mem.writeInt(u32, b[off..][0..4], v, .big); +} +inline fn rdIp(b: []const u8, off: usize) Ip4 { + return b[off..][0..4].*; +} +inline fn wrIp(b: []u8, off: usize, v: Ip4) void { + b[off..][0..4].* = v; +} +inline fn rdMac(b: []const u8, off: usize) Mac { + return b[off..][0..6].*; +} +inline fn wrMac(b: []u8, off: usize, v: Mac) void { + b[off..][0..6].* = v; +} + +// ============================================================================ header offsets +// +// Byte offsets rather than packed structs. `extern struct` would need `align(1)` on every field and +// a byte-swap on every access on this little-endian part, and the offsets are what the RFCs and +// lwIP's headers actually state, so this is the form that can be checked against them by eye. + +/// lwIP `prot/ethernet.h:76-83` (`struct eth_hdr`). +const eth = struct { + const dst = 0; + const src = 6; + const ethertype = 12; +}; + +/// lwIP `prot/etharp.h:86-96` (`struct etharp_hdr`), `SIZEOF_ETHARP_HDR` 28 at `:102`. +const arp = struct { + const hwtype = 0; + const proto = 2; + const hwlen = 4; + const protolen = 5; + const opcode = 6; + const sha = 8; // sender hardware address + const spa = 14; // sender protocol address + const tha = 18; // target hardware address + const tpa = 24; // target protocol address + const len = 28; + + /// lwIP `prot/iana.h:54` (`LWIP_IANA_HWTYPE_ETHERNET`). + const hwtype_ethernet: u16 = 1; + /// lwIP `prot/etharp.h:105-108` (`enum etharp_opcode`). + const op_request: u16 = 1; + const op_reply: u16 = 2; +}; + +/// lwIP `prot/ip4.h:73-97` (`struct ip_hdr`), `IP_HLEN` 20 at `:64`. +const ip4 = struct { + const v_hl = 0; + const tos = 1; + const total_len = 2; + const id = 4; + const frag = 6; + const ttl = 8; + const proto = 9; + const chksum = 10; + const src = 12; + const dst = 16; + const hlen = 20; + + /// lwIP `prot/ip4.h:84-87`. + const flag_df: u16 = 0x4000; + const flag_mf: u16 = 0x2000; + const offset_mask: u16 = 0x1fff; +}; + +/// lwIP `prot/icmp.h:89-95` (`struct icmp_echo_hdr`). +const icmp = struct { + const type_ = 0; + const code = 1; + const chksum = 2; + const id = 4; + const seq = 6; + const hlen = 8; + + /// lwIP `prot/icmp.h:46,50`. + const echo_reply: u8 = 0; + const echo_request: u8 = 8; +}; + +/// lwIP `prot/udp.h:53-58` (`struct udp_hdr`), `UDP_HLEN` 8 at `:46`. +const udp = struct { + const src_port = 0; + const dst_port = 2; + const len = 4; + const chksum = 6; + const hlen = 8; +}; + +/// lwIP `prot/tcp.h:56-65` (`struct tcp_hdr`), `TCP_HLEN` 20 at `:47`. +const tcp = struct { + const src_port = 0; + const dst_port = 2; + const seq = 4; + const ack = 8; + /// Top four bits are the header length in 32-bit words; the low six are the flags. + /// lwIP `prot/tcp.h:85-87`. + const hdrlen_flags = 12; + const window = 14; + const chksum = 16; + const urgent = 18; + const hlen = 20; + + /// lwIP `prot/tcp.h:72-81`. + const fin: u8 = 0x01; + const syn: u8 = 0x02; + const rst: u8 = 0x04; + const psh: u8 = 0x08; + const ack_f: u8 = 0x10; + const urg: u8 = 0x20; + + /// RFC 793 3.1: kind 2, length 4, then the 16-bit MSS. + const opt_end: u8 = 0; + const opt_nop: u8 = 1; + const opt_mss: u8 = 2; +}; + +/// lwIP `prot/dhcp.h:50-91` (`struct dhcp_msg`) and `:51-56` for the offsets named there. +const dhcp = struct { + const op = 0; + const htype = 1; + const hlen = 2; + const hops = 3; + const xid = 4; + const secs = 8; + const flags = 10; + const ciaddr = 12; + const yiaddr = 16; + const siaddr = 20; + const giaddr = 24; + const chaddr = 28; + const sname = 44; // DHCP_SNAME_OFS + const file = 108; // DHCP_FILE_OFS + const cookie = 236; // DHCP_MSG_LEN + const options = 240; // DHCP_OPTIONS_OFS = DHCP_MSG_LEN + 4 + + /// lwIP `prot/dhcp.h:116-117`. + const bootrequest: u8 = 1; + const bootreply: u8 = 2; + /// lwIP `prot/dhcp.h:120-127`. + const discover: u8 = 1; + const offer: u8 = 2; + const request: u8 = 3; + const ack: u8 = 5; + const nak: u8 = 6; + /// lwIP `prot/dhcp.h:129`. + const magic_cookie: u32 = 0x63825363; + /// RFC 2131 figure 2: the top bit of `flags` asks the server to broadcast its reply. + const flag_broadcast: u16 = 0x8000; + + /// lwIP `prot/dhcp.h:134-165`. Only the ones this client uses. + const opt_pad: u8 = 0; + const opt_subnet_mask: u8 = 1; + const opt_router: u8 = 3; + const opt_dns: u8 = 6; + const opt_hostname: u8 = 12; + const opt_requested_ip: u8 = 50; + const opt_lease_time: u8 = 51; + const opt_overload: u8 = 52; + const opt_msg_type: u8 = 53; + const opt_server_id: u8 = 54; + const opt_param_list: u8 = 55; + const opt_max_msg_size: u8 = 57; + const opt_t1: u8 = 58; + const opt_t2: u8 = 59; + const opt_end: u8 = 255; + + /// lwIP `prot/iana.h:66-68`. + const server_port: u16 = 67; + const client_port: u16 = 68; +}; + +/// RFC 1035 4.1. There is no lwIP reference for this one: lwIP's resolver is `core/dns.c`, which +/// builds the same header out of its own `struct dns_hdr` (`core/dns.c:180-190`) - the offsets +/// below are the RFC's, and `dns.c` is only a cross-check. +const dns = struct { + // 4.1.1 header, six 16-bit fields. + const id = 0; + const flags = 2; + const qdcount = 4; + const ancount = 6; + const nscount = 8; + const arcount = 10; + const hlen = 12; + + /// 4.1.1: QR is the top bit of `flags`, RD is bit 8, RCODE the bottom four bits. + const flag_qr: u16 = 0x8000; + const flag_rd: u16 = 0x0100; + const rcode_mask: u16 = 0x000f; + /// RCODE 3, "name error": the name authoritatively does not exist. RFC 1035 4.1.1. + const rcode_name_error: u16 = 3; + + /// 4.1.4: the two top bits of a length byte set means the rest is a 14-bit offset. + const ptr_mask: u8 = 0xc0; + /// 2.3.4: a label is at most 63 bytes, which is also why 0x40 and 0x80 are free to be flags. + const label_max: u8 = 63; + + /// 3.2.2 TYPE and 3.2.4 CLASS. Only the two this stack looks at, plus CNAME, which is not + /// followed but must be stepped over: a name behind a CNAME chain answers with the chain and + /// the A record together, and a resolver that stops at the first record finds the CNAME. + const type_a: u16 = 1; + const type_cname: u16 = 5; + const class_in: u16 = 1; + + /// 3.2.1: TYPE(2) CLASS(2) TTL(4) RDLENGTH(2) after the name. + const rr_fixed = 10; + + /// lwIP `prot/iana.h:64` (`LWIP_IANA_PORT_DNS`). + const port: u16 = 53; +}; + +// ============================================================================== ARP cache + +const ArpEntry = struct { + ip: Ip4 = ip_any, + mac: Mac = @splat(0), + /// `tick`'s clock at the last hit or update. Zero means the entry is empty. + stamp_ms: u64 = 0, + + inline fn valid(self: ArpEntry) bool { + return self.stamp_ms != 0; + } +}; + +/// Entries older than this are treated as absent and re-resolved. lwIP's default is 300 s +/// (`ARP_TMR_INTERVAL` 1000 ms x `ARP_MAXAGE` 300, `core/ipv4/etharp.c`); the same here. +const arp_max_age_ms: u64 = 300_000; +/// How often an unanswered ARP request is repeated while `httpGet` waits for a MAC address. +const arp_retry_ms: u64 = 1000; +/// ARP requests sent for one destination before `httpGet` gives up with `error.HostUnreachable`. +const arp_max_tries: u8 = 5; + +// ================================================================================ DHCP state + +pub const DhcpState = enum { + /// `dhcpStart` has not been called, or `setStatic` has taken over. + off, + /// DISCOVER sent, waiting for an OFFER. + selecting, + /// REQUEST sent, waiting for an ACK. + requesting, + /// Bound, lease held, T1 not yet reached. + bound, + /// Past T1: unicast REQUEST to the server that granted the lease. + renewing, + /// Past T2: broadcast REQUEST to any server. + rebinding, +}; + +const Dhcp = struct { + state: DhcpState = .off, + xid: u32 = 0, + /// The address the server offered, held between OFFER and ACK. + offered: Ip4 = ip_any, + /// Option 54 from the OFFER, echoed in the REQUEST and unicast to when renewing. + server: Ip4 = ip_any, + /// Option 51, seconds. `0xffff_ffff` is an infinite lease (RFC 2131 3.3). + lease_s: u32 = 0, + /// Absolute deadlines derived from the lease at bind time, in `tick`'s milliseconds. + t1_ms: u64 = 0, + t2_ms: u64 = 0, + expire_ms: u64 = 0, + /// When the next DISCOVER/REQUEST retransmission is due, and how many have gone out. + retry_ms: u64 = 0, + tries: u8 = 0, + /// `tick`'s clock when acquisition began, for the `secs` field. + started_ms: u64 = 0, +}; + +// ================================================================================ TCP state + +pub const TcpState = enum { + closed, + /// Waiting for the peer's MAC address before the SYN can be built. + arp_wait, + syn_sent, + established, + /// Our FIN is sent; the peer has not FINed. + fin_wait_1, + fin_wait_2, + /// The peer FINed first and we have replied with our own FIN. + last_ack, + time_wait, +}; + +const Tcp = struct { + state: TcpState = .closed, + + peer_ip: Ip4 = ip_any, + peer_port: u16 = 0, + local_port: u16 = 0, + + /// Initial send sequence number. The SYN occupies `iss`; request data occupies + /// `iss+1 .. iss+1+tx_len`; a FIN occupies `iss+1+tx_len`. Every offset in this struct is + /// derived from that one layout, which is why there is no separate "unacked offset". + iss: u32 = 0, + /// Oldest sequence number not yet acknowledged by the peer. + snd_una: u32 = 0, + /// Next sequence number to send. + snd_nxt: u32 = 0, + /// The peer's advertised window. + snd_wnd: u32 = 0, + /// The peer's MSS, from its SYN's option or RFC 1122's default. + snd_mss: u16 = tcp_default_mss, + /// Set once a FIN has been queued behind the request data. + fin_queued: bool = false, + /// Set once the peer's FIN has been received in order. Receiving it does not by itself move + /// `state`, so that `tcpSendData` stays the only thing that changes it. + peer_fin: bool = false, + + /// Next sequence number expected from the peer. + rcv_nxt: u32 = 0, + + /// Retransmission deadline in `tick`'s milliseconds, and the current timeout. `rto_ms` doubles + /// on every retransmission and is never reduced by an RTT measurement, because this stack + /// takes none: with a single segment in flight and a fixed backoff there is nothing an RTT + /// estimator would change except the first timeout, and 1 s is already RFC 6298's answer for + /// that case. + rto_deadline_ms: u64 = 0, + rto_ms: u32 = tcp_rto_initial_ms, + retries: u8 = 0, + /// When TIME_WAIT ends. + close_deadline_ms: u64 = 0, + + /// The request bytes, held for retransmission until acknowledged. + tx: [tcp_tx_max]u8 = undefined, + tx_len: usize = 0, + + /// Sequence number of the first byte of `tx`. + inline fn dataStart(self: Tcp) u32 { + return self.iss +% 1; + } + /// Sequence number one past the last byte of `tx`. + inline fn dataEnd(self: Tcp) u32 { + return self.iss +% 1 +% @as(u32, @intCast(self.tx_len)); + } +}; + +/// Sequence-number comparison. TCP sequence numbers wrap, so they are compared by the sign of the +/// difference and never by `<`. RFC 1982 serial arithmetic; the classic bug this avoids is a +/// connection that stalls forever once the sequence space crosses 2^32. +inline fn seqLt(a: u32, b: u32) bool { + return @as(i32, @bitCast(a -% b)) < 0; +} +inline fn seqLe(a: u32, b: u32) bool { + return @as(i32, @bitCast(a -% b)) <= 0; +} +inline fn seqGt(a: u32, b: u32) bool { + return seqLt(b, a); +} +inline fn seqGe(a: u32, b: u32) bool { + return seqLe(b, a); +} + +// =============================================================================== HTTP state + +pub const HttpError = error{ + /// The request is in flight. Call `tick`, feed frames to `onFrame`, and call `httpGet` again + /// with the same arguments. This is the only "error" a healthy request returns. + WouldBlock, + /// `httpGet` was called with different arguments while a request was in flight. + Busy, + /// The peer's MAC address could not be resolved. + HostUnreachable, + /// The peer sent RST. + ConnectionReset, + /// The peer FINed or vanished before the body was complete. + ConnectionClosed, + /// Retransmissions exhausted. + TimedOut, + /// The status line, the headers, or a chunked body's trailer section exceeded its budget + /// (`http_head_max`, `http_framing_max`). + HttpHeadersTooLong, + /// The status line was not `HTTP/1.x SSS`. + HttpMalformed, + /// A chunked body's framing was not RFC 7230 4.1: a size with no hex digits, a size that + /// overflows `usize`, or a CRLF that was not where the grammar puts it. Distinct from + /// `HttpMalformed` because the two point at different halves of the response, and on a board + /// with one UART the error name is the whole diagnosis. + HttpChunkMalformed, + /// `Transfer-Encoding` was present and was neither `identity` nor `chunked`. + UnsupportedTransferEncoding, + /// The body did not fit in the caller's `out` buffer. + StreamTooLong, + /// The request line and headers did not fit in `tcp_tx_max`, or `path` is unusable. + RequestTooLong, + /// `httpGet` was called before the stack had an address. + NoAddress, +}; + +const HttpPhase = enum { idle, head, body, complete, failed }; + +const Http = struct { + phase: HttpPhase = .idle, + /// Valid when `phase == .failed`. + err: HttpError = error.WouldBlock, + + /// The caller's output buffer, borrowed for the duration of the request. Recorded rather than + /// copied, so the caller must not move or resize it between `httpGet` calls; the identity check + /// in `httpGet` catches the common way of getting that wrong. + out: []u8 = &.{}, + out_len: usize = 0, + + /// The request being served, kept so a re-entrant `httpGet` can be told apart from a new one. + /// The hash covers the path *and* the `Host:` name, which is what makes two requests to the + /// same address for the same path but different virtual hosts distinguishable - and they must + /// be, or the second silently rides on the first's connection. A hash rather than the strings + /// themselves because `Stack` has a 4 KiB budget and the strings are the caller's, alive for + /// the duration of the call only. + req_host: Ip4 = ip_any, + req_port: u16 = 0, + req_hash: u64 = 0, + + /// Status line and headers, accumulated until the blank line. + head: [http_head_max]u8 = undefined, + head_len: usize = 0, + + status: u16 = 0, + /// `null` means the response had no usable `Content-Length`, so the body ends at the peer's + /// FIN - or, when `chunked`, at the zero-length chunk. + content_length: ?usize = null, + + // ------------------------------------------------------- RFC 7230 4.1 chunked decoding + // + // Four fields hold the whole position in the chunked grammar, because a segment boundary may + // fall between any two bytes of it and the decoder has to resume from exactly here. + + /// `Transfer-Encoding: chunked` was in force on this response. + chunked: bool = false, + chunk: ChunkState = .size, + /// In `.size`, the hexadecimal size accumulated so far; in `.data`, the bytes of this chunk + /// still to come. The two are the same number, which is why one field serves both: the size + /// read is the count remaining the instant the header ends. + chunk_left: usize = 0, + /// At least one hex digit has been seen in the size being read. RFC 7230 4.1 is `1*HEXDIG`, + /// so an empty size is malformed - and without this flag a stray CRLF reads as a chunk of + /// length zero, which is the terminator, which ends the body early and looks like success. + chunk_digit: bool = false, + /// Framing bytes consumed in the extension or trailer section now being skipped, against + /// `http_framing_max`. + chunk_skip: u16 = 0, +}; + +/// Where the chunked decoder is in RFC 7230 4.1's grammar: +/// +/// chunked-body = *chunk last-chunk trailer-part CRLF +/// chunk = chunk-size [ chunk-ext ] CRLF chunk-data CRLF +/// last-chunk = 1*("0") [ chunk-ext ] CRLF +/// +/// Every terminal in that grammar that can be split by a segment boundary is a state, including +/// the two halves of each CRLF. That is not pedantry: a 1,460-byte segment ends wherever the +/// server's writes happen to end, and "the CR arrived and the LF did not" is a case that happens. +const ChunkState = enum { + /// Reading hex digits of `chunk-size`. + size, + /// A `;` was seen: skipping `chunk-ext` to the CR that ends the header. + ext, + /// The CR of the chunk header is in; its LF must follow. + size_lf, + /// Copying `chunk_left` more bytes of `chunk-data` into the caller's `out`. + data, + /// The data is in; the CR of the CRLF that closes the chunk must follow. + data_cr, + /// ...and its LF. + data_lf, + /// At the first byte of a trailer line - or of the CRLF that ends the whole body. + trailer, + /// Inside a trailer line, skipping to its CR. + trailer_line, + /// The LF of a trailer line. + trailer_lf, + /// The LF of the final empty line. The response is complete after it, and not before. + end_lf, +}; + +// ================================================================================ DNS state + +pub const DnsError = error{ + /// The query is in flight. Call `tick`, feed frames to `onFrame`, and call `resolve` again + /// with the same name. This is the only "error" a healthy query returns. + WouldBlock, + /// `resolve` was called with a different name while a query was in flight. One query is + /// outstanding at a time; the caller must finish or abandon the first. + Busy, + /// `resolve` was called before the stack had an address of its own to send from. + NoAddress, + /// No resolver: DHCP supplied none and `setDnsServer` was not called. + NoDnsServer, + /// The name was empty, had an empty label, or had a label over 63 bytes. RFC 1035 2.3.4. + NameInvalid, + /// The name was longer than `dns_name_max`. + NameTooLong, + /// `dns_backoff_ms.len` queries went out and nothing came back. + TimedOut, + /// The server said the name does not exist (RCODE 3), or answered with no A record in it - + /// a CNAME chain leading nowhere, or an AAAA-only name. Both mean the same thing to a stack + /// that speaks IPv4 only. + NameNotFound, + /// The server answered with a non-zero RCODE other than name-error: SERVFAIL, REFUSED. + DnsRefused, + /// A response that matched the id and the question could not be parsed: a name that runs off + /// the end, a compression pointer that goes forward or loops, an RDLENGTH past the message. + /// Responses that do *not* match the id and question are ignored rather than reported, so + /// this is the server or an attacker who already guessed both, never stray traffic. + DnsMalformed, +}; + +const DnsPhase = enum { idle, waiting, done, failed }; + +const DnsQuery = struct { + phase: DnsPhase = .idle, + /// Valid when `phase == .failed`. + err: DnsError = error.WouldBlock, + + /// The question, in RFC 1035 4.1.2 wire form, root label included. Held rather than + /// re-encoded because it is needed in three places: to build each retransmission from `tick`, + /// to compare against the question echoed in a response, and to tell a re-entrant `resolve` + /// from a new one. Comparing the encoded form is what makes the last two exact. + qname: [dns_qname_max]u8 = undefined, + qname_len: u8 = 0, + + /// The transaction id, held across retransmissions so a slow first answer still matches. + id: u16 = 0, + /// The ephemeral source port, redrawn per query. Together with `id` that is 32 bits an + /// off-path spoofer has to guess, which is the whole of what plain DNS offers. + local_port: u16 = 0, + + tries: u8 = 0, + retry_ms: u64 = 0, + + /// Valid when `phase == .done`. + result: Ip4 = ip_any, +}; + +// ================================================================================= counters +// +// Not statistics for their own sake: the first hardware bring-up of this stack will be a board that +// either answers a ping or does not, with no debugger and one UART. These are what turns "nothing +// happens" into "1,204 frames arrived, 1,204 were dropped, and the checksum counter is zero", which +// says the frames are not for us rather than that the checksum code is broken. + +pub const Counters = struct { + rx_frames: u32 = 0, + rx_dropped: u32 = 0, + tx_frames: u32 = 0, + tx_dropped: u32 = 0, + arp_rx: u32 = 0, + arp_tx: u32 = 0, + icmp_echo: u32 = 0, + udp_rx: u32 = 0, + dhcp_rx: u32 = 0, + dhcp_tx: u32 = 0, + tcp_rx: u32 = 0, + tcp_tx: u32 = 0, + tcp_retx: u32 = 0, + tcp_rst_rx: u32 = 0, + /// Queries sent, retransmissions among them, and responses that matched the outstanding + /// query's id and question. `dns_tx > dns_rx` with `dns_retx` climbing is a resolver that is + /// not answering; `dns_rx == 0` with `udp_rx` climbing is an answer arriving and being + /// rejected, which is a different bug in a different place. + dns_tx: u32 = 0, + dns_retx: u32 = 0, + dns_rx: u32 = 0, + /// Frames discarded because a checksum did not verify. A non-zero value here with a working + /// link means a bug in this file or a broken SDIO transfer, not a network problem. + checksum_bad: u32 = 0, +}; + +// ==================================================================================== Stack + +pub const Stack = struct { + // ------------------------------------------------------------------ identity and route + mac: Mac, + /// `null` until DHCP binds or `setStatic` is called. + addr: ?Ip4 = null, + mask: Ip4 = ip_any, + gw: Ip4 = ip_any, + /// The resolver, from DHCP option 6 or `setDnsServer`. `null` means nothing to ask. + dns: ?Ip4 = null, + + /// Where frames go. Called synchronously from `onFrame`, `tick` and `httpGet`; the slice is + /// borrowed for the duration of the call and must be copied if the transport needs it later. + /// + /// Note the absence of a context pointer: the interface this slice implements specifies + /// `*const fn ([]const u8) void`, so a callee needing state has to reach it some other way. + send: *const fn (frame: []const u8) void, + + // ------------------------------------------------------------------------------ clock + /// The last value handed to `tick`. `onFrame` needs a timestamp for the ARP cache and takes it + /// from here rather than reading a clock, which is what keeps this file free of any hardware + /// dependency at all. + now_ms: u64 = 0, + + // ------------------------------------------------------------------------------ state + arp_cache: [arp_cache_len]ArpEntry = @splat(.{}), + /// Pending ARP resolution for the TCP peer: deadline, tries. + arp_retry_ms: u64 = 0, + arp_tries: u8 = 0, + + dhcp: Dhcp = .{}, + tcp: Tcp = .{}, + http: Http = .{}, + /// The one outstanding DNS query. Named `query` and not `dns`, which is the server's address. + query: DnsQuery = .{}, + counters: Counters = .{}, + + /// IPv4 identification field. Incremented per datagram. Nothing here fragments, so this only + /// has to be non-constant for the benefit of middleboxes and packet captures. + ip_id: u16 = 0, + /// Mixed into transaction ids, initial sequence numbers and ephemeral ports. There is no + /// hardware RNG in this file's reach, so this is seeded from the MAC and stirred by every + /// `tick` value observed - which for the two uses here (not colliding with a previous + /// incarnation of the same connection, and not matching a stale DHCP reply) is sufficient. + /// It is emphatically *not* a source of security-relevant randomness. + entropy: u64, + + /// The single transmit staging buffer. Every frame this stack sends is built here and handed to + /// `send` before the next one starts, so one is enough - and `send` is documented as borrowing. + tx: [frame_max]u8 = undefined, + + /// The total static footprint of one `Stack`, asserted so the number in the report cannot rot. + pub const footprint = @sizeOf(Stack); + + // =========================================================================== lifecycle + + /// A single struct-literal return, deliberately: result-location semantics then construct the + /// buffers in the caller's storage instead of memcpy-ing several kilobytes off a stack that is + /// 8 KB by default on this target. + pub fn init(mac: [6]u8, send: *const fn (frame: []const u8) void) Stack { + return .{ + .mac = mac, + .send = send, + .entropy = std.hash.Wyhash.hash(0x4200_cafe, &mac), + }; + } + + /// Stir and draw. Not random; see `entropy`. + fn draw(self: *Stack) u32 { + self.entropy = self.entropy *% 6364136223846793005 +% 1442695040888963407; + return @truncate(self.entropy >> 32); + } + + // ============================================================================= address + + /// The configured address, or `null` if there is none yet. + pub fn ip(self: *Stack) ?[4]u8 { + return self.addr; + } + + pub fn netmask(self: *Stack) Ip4 { + return self.mask; + } + + pub fn gateway(self: *Stack) Ip4 { + return self.gw; + } + + /// The resolver `resolve` will ask: the first server DHCP offered (option 6), or whatever + /// `setDnsServer` last set. `null` means `resolve` will answer `error.NoDnsServer`. + pub fn dnsServer(self: *Stack) ?Ip4 { + return self.dns; + } + + /// Override the resolver. Only needed on a network whose DHCP server offers none, or when + /// configuring statically: the ordinary path is a lease that carries option 6, which + /// `dhcpBind` already stores, and a caller that does nothing gets that. + /// + /// Any query in flight is abandoned: it was addressed to the old server and its answer would + /// now be rejected as coming from the wrong source. + pub fn setDnsServer(self: *Stack, addr: Ip4) void { + self.dns = addr; + self.query.phase = .idle; + } + + pub fn dhcpState(self: *Stack) DhcpState { + return self.dhcp.state; + } + + pub fn tcpState(self: *Stack) TcpState { + return self.tcp.state; + } + + /// The status code of the last response whose head was parsed. Zero before that. + pub fn httpStatus(self: *Stack) u16 { + return self.http.status; + } + + /// Configure statically and stop any DHCP activity. This is the path the first hardware test + /// takes: it makes the board reachable without a working DHCP client, so an ARP or ping + /// failure means the SDIO transport or the association is wrong rather than this file. + pub fn setStatic(self: *Stack, addr: [4]u8, mask: [4]u8, gw: [4]u8) void { + self.dhcp = .{}; + self.addr = addr; + self.mask = mask; + self.gw = gw; + // Any query in flight was sent from the old address, so its answer is addressed to a + // station that no longer exists. `dns` itself is left alone: a resolver learnt from a + // previous lease is still the right one to ask on the same wire. + self.query.phase = .idle; + self.announce(); + } + + /// Is this address ours, or one everybody on the wire is meant to hear? + fn forUs(self: *Stack, dst: Ip4) bool { + if (std.mem.eql(u8, &dst, &ip_broadcast)) return true; + const a = self.addr orelse return false; + if (std.mem.eql(u8, &dst, &a)) return true; + // Subnet broadcast: host bits all ones. + var i: usize = 0; + while (i < 4) : (i += 1) { + if (dst[i] | self.mask[i] != 0xff) return false; + if (dst[i] & self.mask[i] != a[i] & self.mask[i]) return false; + } + return true; + } + + fn onLink(self: *Stack, dst: Ip4) bool { + const a = self.addr orelse return true; // unconfigured: everything is a direct neighbour + var i: usize = 0; + while (i < 4) : (i += 1) { + if ((dst[i] ^ a[i]) & self.mask[i] != 0) return false; + } + return true; + } + + // ================================================================== frame construction + + fn emitFrame(self: *Stack, len: usize) void { + if (len > frame_max) { + self.counters.tx_dropped += 1; + return; + } + // Ethernet's 60-byte minimum (64 with FCS) is padded by the MAC, and ESP-Hosted's slave + // hands the frame to the C6's Wi-Fi MAC, which does the same. Nothing is padded here. + self.counters.tx_frames += 1; + self.send(self.tx[0..len]); + } + + fn ethHeader(self: *Stack, dst: Mac, ethertype: EtherType) void { + wrMac(&self.tx, eth.dst, dst); + wrMac(&self.tx, eth.src, self.mac); + wr16(&self.tx, eth.ethertype, @intFromEnum(ethertype)); + } + + /// Build and send an IPv4 datagram whose payload the caller has already written to + /// `self.tx[eth_hlen + ip4.hlen ..]`. Returns false if the destination's MAC is unknown, in + /// which case an ARP request has been sent and the datagram is dropped. + /// + /// Dropping rather than queueing is lwIP's `ETHARP_SUPPORT_STATIC_ENTRIES`-less behaviour minus + /// its one-packet queue (`core/ipv4/etharp.c`, `etharp_query`). Nothing here needs the queue: + /// DHCP is broadcast, ICMP replies go to a peer whose MAC just arrived in the request, and TCP + /// resolves the peer before the SYN is built (`TcpState.arp_wait`). + fn emitIp(self: *Stack, src: Ip4, dst: Ip4, proto: Protocol, payload_len: usize) bool { + assert(payload_len <= mtu - ip4.hlen); + const total: u16 = @intCast(ip4.hlen + payload_len); + + const dst_mac = self.routeMac(dst) orelse { + self.counters.tx_dropped += 1; + return false; + }; + self.ethHeader(dst_mac, .ip4); + + const h = self.tx[eth_hlen..][0..ip4.hlen]; + h[ip4.v_hl] = 0x45; // IPv4, 5 words of header, no options + h[ip4.tos] = 0; + wr16(h, ip4.total_len, total); + wr16(h, ip4.id, self.ip_id); + self.ip_id +%= 1; + // DF set: this stack neither fragments what it sends nor reassembles what it receives, so + // saying so is more useful than letting a router fragment a datagram we cannot rebuild. + wr16(h, ip4.frag, ip4.flag_df); + h[ip4.ttl] = 64; // RFC 1122 3.2.1.7 recommends 64 + h[ip4.proto] = @intFromEnum(proto); + wr16(h, ip4.chksum, 0); + wrIp(h, ip4.src, src); + wrIp(h, ip4.dst, dst); + wr16(h, ip4.chksum, checksum(h)); + + self.emitFrame(eth_hlen + total); + return true; + } + + /// The MAC a datagram for `dst` must be sent to: broadcast for a broadcast address, the peer + /// itself if it is on-link, otherwise the gateway. `null` means unresolved, and an ARP request + /// has been sent. + fn routeMac(self: *Stack, dst: Ip4) ?Mac { + if (std.mem.eql(u8, &dst, &ip_broadcast)) return mac_broadcast; + if (self.addr != null) { + // Subnet broadcast. + var all_ones = true; + var i: usize = 0; + while (i < 4) : (i += 1) { + if (dst[i] | self.mask[i] != 0xff) all_ones = false; + } + if (all_ones and self.onLink(dst)) return mac_broadcast; + } + const next = if (self.onLink(dst)) dst else self.gw; + if (self.arpLookup(next)) |m| return m; + self.arpRequest(next); + return null; + } + + // ================================================================================= ARP + + /// An entry's timestamp is *not* refreshed by a lookup, only by an ARP packet from that host. + /// Refreshing on use looks like a cheap optimisation and is a real bug: an entry kept alive by + /// our own traffic is never re-resolved, so a gateway whose MAC changes - VRRP failover, a + /// replaced router, a roam to a different AP with a different BSSID-derived address - is never + /// noticed, and every frame goes to a MAC that no longer answers. Ageing out after + /// `arp_max_age_ms` of no ARP traffic from that host costs one dropped segment and a + /// retransmission; getting it wrong costs the connection. + fn arpLookup(self: *Stack, target: Ip4) ?Mac { + for (&self.arp_cache) |*e| { + if (!e.valid()) continue; + if (self.now_ms -% e.stamp_ms > arp_max_age_ms) { + e.stamp_ms = 0; + continue; + } + if (std.mem.eql(u8, &e.ip, &target)) return e.mac; + } + return null; + } + + /// Insert or refresh. `insert` false means "update only if already known", which is how a + /// four-entry cache survives a busy /24: every ARP request on the segment is a broadcast, and a + /// cache that admitted all of them would evict the gateway within seconds. lwIP draws the same + /// line with `ETHARP_FLAG_TRY_HARD` (`core/ipv4/etharp.c`, `etharp_update_arp_entry`). + fn arpStore(self: *Stack, target: Ip4, hw: Mac, insert: bool) void { + if (std.mem.eql(u8, &target, &ip_any)) return; + if (std.mem.eql(u8, &target, &ip_broadcast)) return; + for (&self.arp_cache) |*e| { + if (e.valid() and std.mem.eql(u8, &e.ip, &target)) { + e.mac = hw; + e.stamp_ms = self.now_ms; + return; + } + } + if (!insert) return; + // Free slot, else the least recently used. + var victim: *ArpEntry = &self.arp_cache[0]; + for (&self.arp_cache) |*e| { + if (!e.valid()) { + victim = e; + break; + } + if (e.stamp_ms < victim.stamp_ms) victim = e; + } + victim.* = .{ .ip = target, .mac = hw, .stamp_ms = self.now_ms }; + } + + fn arpEmit(self: *Stack, opcode: u16, target_ip: Ip4, target_mac: Mac, dst_mac: Mac, spa: Ip4) void { + self.ethHeader(dst_mac, .arp); + const h = self.tx[eth_hlen..][0..arp.len]; + wr16(h, arp.hwtype, arp.hwtype_ethernet); + wr16(h, arp.proto, @intFromEnum(EtherType.ip4)); + h[arp.hwlen] = 6; + h[arp.protolen] = 4; + wr16(h, arp.opcode, opcode); + wrMac(h, arp.sha, self.mac); + wrIp(h, arp.spa, spa); + wrMac(h, arp.tha, target_mac); + wrIp(h, arp.tpa, target_ip); + self.counters.arp_tx += 1; + self.emitFrame(eth_hlen + arp.len); + } + + fn arpRequest(self: *Stack, target: Ip4) void { + // RFC 826: the target hardware address of a request is "don't care"; zero is conventional. + self.arpEmit(arp.op_request, target, @splat(0), mac_broadcast, self.addr orelse ip_any); + } + + /// Gratuitous ARP: a broadcast request for our own address, which every listener treats as + /// "this MAC now owns this IP". Sent when an address is acquired, so the gateway and the AP + /// learn us without waiting to need us. RFC 5227 2.3. + fn announce(self: *Stack) void { + const a = self.addr orelse return; + self.arpEmit(arp.op_request, a, @splat(0), mac_broadcast, a); + } + + fn arpInput(self: *Stack, body: []const u8) void { + if (body.len < arp.len) { + self.counters.rx_dropped += 1; + return; + } + // RFC 826 "Packet Reception", exactly the four checks lwIP makes at + // `core/ipv4/etharp.c:656-659`. + if (rd16(body, arp.hwtype) != arp.hwtype_ethernet or + rd16(body, arp.proto) != @intFromEnum(EtherType.ip4) or + body[arp.hwlen] != 6 or body[arp.protolen] != 4) + { + self.counters.rx_dropped += 1; + return; + } + self.counters.arp_rx += 1; + + const spa = rdIp(body, arp.spa); + const sha = rdMac(body, arp.sha); + const tpa = rdIp(body, arp.tpa); + const for_us = if (self.addr) |a| std.mem.eql(u8, &tpa, &a) else false; + + // Learn the sender. Admitted to a free slot only when the packet was addressed to us - + // either a request we must answer or the reply to a request we sent. + self.arpStore(spa, sha, for_us); + + if (rd16(body, arp.opcode) == arp.op_request and for_us) { + // A reply goes back to the requester, not to the broadcast address. + self.arpEmit(arp.op_reply, spa, sha, sha, self.addr.?); + } + } + + // =============================================================================== input + + /// A received Ethernet frame. Everything this stack does in response happens before this + /// returns, including any frame it sends. + pub fn onFrame(self: *Stack, frame: []const u8) void { + self.counters.rx_frames += 1; + if (frame.len < eth_hlen or frame.len > frame_max) { + self.counters.rx_dropped += 1; + return; + } + const dst = rdMac(frame, eth.dst); + // The C6's MAC filter should already have done this, but a promiscuous or misconfigured + // transport would otherwise have this stack answering ARP for other stations. + if (!std.mem.eql(u8, &dst, &self.mac) and !std.mem.eql(u8, &dst, &mac_broadcast)) { + self.counters.rx_dropped += 1; + return; + } + const body = frame[eth_hlen..]; + switch (@as(EtherType, @enumFromInt(rd16(frame, eth.ethertype)))) { + .arp => self.arpInput(body), + .ip4 => self.ip4Input(body), + // .vlan lands here: an 802.1Q tag would need the 4-byte shim skipped and the real + // ethertype read from behind it. Nothing on this board tags frames, so it is dropped + // rather than half-handled. + else => self.counters.rx_dropped += 1, + } + } + + fn ip4Input(self: *Stack, body: []const u8) void { + if (body.len < ip4.hlen) { + self.counters.rx_dropped += 1; + return; + } + if (body[ip4.v_hl] >> 4 != 4) { + self.counters.rx_dropped += 1; + return; + } + const hlen = @as(usize, body[ip4.v_hl] & 0x0f) * 4; + if (hlen < ip4.hlen or hlen > body.len) { + self.counters.rx_dropped += 1; + return; + } + if (checksum(body[0..hlen]) != 0) { + self.counters.checksum_bad += 1; + return; + } + const total = rd16(body, ip4.total_len); + if (total < hlen or total > body.len) { + // Shorter than claimed: truncated. Longer than claimed happens legitimately - a + // 60-byte minimum-length Ethernet frame padding a 28-byte datagram - and is handled by + // trusting `total` below, but a frame shorter than its own IP header claims is junk. + self.counters.rx_dropped += 1; + return; + } + const frag = rd16(body, ip4.frag); + if (frag & (ip4.flag_mf | ip4.offset_mask) != 0) { + // A fragment. Reassembly is out of scope, and accepting the first fragment as a whole + // datagram would be worse than dropping it. + self.counters.rx_dropped += 1; + return; + } + + const src = rdIp(body, ip4.src); + const dst = rdIp(body, ip4.dst); + const proto: Protocol = @enumFromInt(body[ip4.proto]); + const payload = body[hlen..total]; + + if (!self.forUs(dst)) { + // One exception, and it is the reason DHCP works at all: a server may unicast its + // OFFER or ACK to the address it is about to grant, which is not yet ours, at a MAC + // that is. RFC 2131 4.1 permits exactly this. So while unbound, UDP is let through to + // the demultiplexer, which will only match the DHCP client port. + const dhcp_pending = self.addr == null and self.dhcp.state != .off; + if (!(dhcp_pending and proto == .udp)) { + self.counters.rx_dropped += 1; + return; + } + } + + switch (proto) { + .icmp => self.icmpInput(src, dst, payload), + .udp => self.udpInput(src, dst, payload), + .tcp => self.tcpInput(src, dst, payload), + else => self.counters.rx_dropped += 1, + } + } + + // ================================================================================ ICMP + + fn icmpInput(self: *Stack, src: Ip4, dst: Ip4, payload: []const u8) void { + if (payload.len < icmp.hlen) { + self.counters.rx_dropped += 1; + return; + } + // ICMP has no pseudo-header (RFC 792): the checksum covers the message alone. + if (checksum(payload) != 0) { + self.counters.checksum_bad += 1; + return; + } + if (payload[icmp.type_] != icmp.echo_request) { + // Destination-unreachable and time-exceeded carry useful information that nothing here + // consumes; a stack with no routing decisions to revise has nothing to do with them. + self.counters.rx_dropped += 1; + return; + } + // A request addressed to the broadcast address is answered from our own address only; a + // reply sourced from a broadcast address is malformed and some hosts treat it as an attack. + if (self.addr == null) return; + if (payload.len > mtu - ip4.hlen) { + // Would need fragmenting to answer. `ping -s 1473` from the development host lands + // here as a fragmented request and is already dropped above; this covers the rest. + self.counters.rx_dropped += 1; + return; + } + _ = dst; + + const out = self.tx[eth_hlen + ip4.hlen ..][0..payload.len]; + @memcpy(out, payload); + out[icmp.type_] = icmp.echo_reply; + out[icmp.code] = 0; + wr16(out, icmp.chksum, 0); + wr16(out, icmp.chksum, checksum(out)); + self.counters.icmp_echo += 1; + _ = self.emitIp(self.addr.?, src, .icmp, payload.len); + } + + // ================================================================================= UDP + + fn udpInput(self: *Stack, src: Ip4, dst: Ip4, payload: []const u8) void { + if (payload.len < udp.hlen) { + self.counters.rx_dropped += 1; + return; + } + const ulen = rd16(payload, udp.len); + if (ulen < udp.hlen or ulen > payload.len) { + self.counters.rx_dropped += 1; + return; + } + const datagram = payload[0..ulen]; + if (!transportChecksumOk(src, dst, .udp, datagram, rd16(datagram, udp.chksum))) { + self.counters.checksum_bad += 1; + return; + } + self.counters.udp_rx += 1; + + const sport = rd16(datagram, udp.src_port); + const dport = rd16(datagram, udp.dst_port); + const data = datagram[udp.hlen..]; + if (dport == dhcp.client_port) { + self.dhcpInput(src, data); + } else if (self.query.phase == .waiting and dport == self.query.local_port) { + self.dnsInput(src, sport, data); + } else { + // No sockets, so nothing else has a port. A real stack would answer with ICMP port + // unreachable; announcing which ports are closed is of no use to this device. + self.counters.rx_dropped += 1; + } + } + + /// Send a UDP datagram. `src` may be `0.0.0.0`, which DHCP needs before it has an address. + fn emitUdp(self: *Stack, src: Ip4, sport: u16, dst: Ip4, dport: u16, data_len: usize) bool { + const seg_len = udp.hlen + data_len; + assert(seg_len <= mtu - ip4.hlen); + const seg = self.tx[eth_hlen + ip4.hlen ..][0..seg_len]; + wr16(seg, udp.src_port, sport); + wr16(seg, udp.dst_port, dport); + wr16(seg, udp.len, @intCast(seg_len)); + wr16(seg, udp.chksum, 0); + wr16(seg, udp.chksum, udpChecksumOnWire(transportChecksum(src, dst, .udp, seg))); + return self.emitIp(src, dst, .udp, seg_len); + } + + // ================================================================================ DHCP + + /// Begin acquiring an address. Idempotent while an acquisition is in progress; a call while + /// bound restarts from DISCOVER. + pub fn dhcpStart(self: *Stack) void { + self.addr = null; + self.mask = ip_any; + self.gw = ip_any; + self.dns = null; + // The resolver is gone with the lease, so anything in flight to it is abandoned rather + // than left to time out against a server this stack no longer believes in. + self.query.phase = .idle; + self.dhcp = .{ + .state = .selecting, + .xid = self.draw(), + .started_ms = self.now_ms, + }; + self.dhcpSend(dhcp.discover); + self.dhcp.tries = 1; + self.dhcp.retry_ms = self.now_ms + dhcp_backoff_ms[0]; + } + + /// Options are appended through this so a length byte can never be written by hand. + const OptWriter = struct { + buf: []u8, + i: usize = 0, + + fn raw(self: *OptWriter, code: u8, value: []const u8) void { + assert(value.len <= 255); + assert(self.i + 2 + value.len <= self.buf.len); + self.buf[self.i] = code; + self.buf[self.i + 1] = @intCast(value.len); + @memcpy(self.buf[self.i + 2 ..][0..value.len], value); + self.i += 2 + value.len; + } + fn byte(self: *OptWriter, code: u8, v: u8) void { + self.raw(code, &[_]u8{v}); + } + fn word(self: *OptWriter, code: u8, v: u16) void { + var t: [2]u8 = undefined; + std.mem.writeInt(u16, &t, v, .big); + self.raw(code, &t); + } + fn address(self: *OptWriter, code: u8, v: Ip4) void { + self.raw(code, &v); + } + fn end(self: *OptWriter) void { + assert(self.i < self.buf.len); + self.buf[self.i] = dhcp.opt_end; + self.i += 1; + } + }; + + /// Build and send one DHCP message. The RFC 2131 4.3.6 table is what decides which fields are + /// set: it is the part of DHCP that servers actually enforce, and getting `ciaddr` or the + /// server identifier wrong produces a NAK rather than an error message. + fn dhcpSend(self: *Stack, kind: u8) void { + const msg = self.tx[eth_hlen + ip4.hlen + udp.hlen ..][0..dhcp_min_msg_len]; + @memset(msg, 0); + + const renewing = self.dhcp.state == .renewing; + const rebinding = self.dhcp.state == .rebinding; + // RENEWING and REBINDING carry the bound address in `ciaddr` and no requested-IP option; + // SELECTING and REQUESTING carry zero and use option 50. lwIP makes the same distinction + // at `core/ipv4/dhcp.c:2026-2030`. + const use_ciaddr = renewing or rebinding; + + msg[dhcp.op] = dhcp.bootrequest; + msg[dhcp.htype] = @intCast(arp.hwtype_ethernet); + msg[dhcp.hlen] = 6; + msg[dhcp.hops] = 0; + wr32(msg, dhcp.xid, self.dhcp.xid); + wr16(msg, dhcp.secs, @intCast(@min(0xffff, (self.now_ms -% self.dhcp.started_ms) / 1000))); + // Ask the server to broadcast its reply. lwIP clears this flag + // (`core/ipv4/dhcp.c:2024-2025`: "we don't need the broadcast flag since we can receive + // unicast traffic before being fully configured"), and so can this stack - `ip4Input` has + // the explicit exemption for it. The flag is set anyway because a broadcast reply is the + // path with the fewest ways to fail on first bring-up: it needs no ARP entry at the server, + // no unicast-to-unconfigured-host handling in the AP, and no exemption in this file. + wr16(msg, dhcp.flags, dhcp.flag_broadcast); + if (use_ciaddr) wrIp(msg, dhcp.ciaddr, self.addr orelse ip_any); + @memcpy(msg[dhcp.chaddr..][0..6], &self.mac); + wr32(msg, dhcp.cookie, dhcp.magic_cookie); + + var o: OptWriter = .{ .buf = msg[dhcp.options..] }; + o.byte(dhcp.opt_msg_type, kind); + // RFC 2131 3.5: the maximum message size we can reassemble. One MTU minus the headers, + // which for this stack is also the largest datagram it can receive at all. + o.word(dhcp.opt_max_msg_size, @intCast(mtu - ip4.hlen - udp.hlen)); + if (kind == dhcp.request and !use_ciaddr) { + o.address(dhcp.opt_requested_ip, self.dhcp.offered); + o.address(dhcp.opt_server_id, self.dhcp.server); + } + // RFC 2131 4.3.6: a REQUEST in RENEWING/REBINDING must not carry a server identifier. + o.raw(dhcp.opt_param_list, &[_]u8{ + dhcp.opt_subnet_mask, + dhcp.opt_router, + dhcp.opt_dns, + dhcp.opt_lease_time, + dhcp.opt_t1, + dhcp.opt_t2, + }); + o.raw(dhcp.opt_hostname, "esp32p4"); + o.end(); + // Everything past the END option stays zero: RFC 2131 4.1 pads with option 0. + + const src = if (use_ciaddr) (self.addr orelse ip_any) else ip_any; + // RENEWING unicasts to the server that granted the lease; every other message is broadcast + // (RFC 2131 4.3.6, 4.4.5). + const dst = if (renewing) self.dhcp.server else ip_broadcast; + self.counters.dhcp_tx += 1; + _ = self.emitUdp(src, dhcp.client_port, dst, dhcp.server_port, dhcp_min_msg_len); + } + + /// One parsed option, or the end of the list. + const Opt = struct { code: u8, value: []const u8 }; + + /// Walk a DHCP option list. Stops at END, at a truncated option, or at the end of the buffer - + /// a malformed length must not walk off the datagram, which is the classic DHCP parser bug. + fn dhcpOption(body: []const u8, want: u8) ?[]const u8 { + if (body.len <= dhcp.options) return null; + var i: usize = dhcp.options; + while (i < body.len) { + const code = body[i]; + if (code == dhcp.opt_end) return null; + if (code == dhcp.opt_pad) { + i += 1; + continue; + } + if (i + 2 > body.len) return null; + const len = body[i + 1]; + if (i + 2 + len > body.len) return null; + if (code == want) return body[i + 2 ..][0..len]; + i += 2 + len; + } + return null; + } + + fn dhcpOptionIp(body: []const u8, want: u8) ?Ip4 { + const v = dhcpOption(body, want) orelse return null; + if (v.len < 4) return null; + return v[0..4].*; + } + + fn dhcpOptionU32(body: []const u8, want: u8) ?u32 { + const v = dhcpOption(body, want) orelse return null; + if (v.len != 4) return null; + return std.mem.readInt(u32, v[0..4], .big); + } + + fn dhcpInput(self: *Stack, src: Ip4, body: []const u8) void { + if (self.dhcp.state == .off) return; + if (body.len < dhcp.options) { + self.counters.rx_dropped += 1; + return; + } + if (body[dhcp.op] != dhcp.bootreply) return; + if (rd32(body, dhcp.cookie) != dhcp.magic_cookie) return; + if (rd32(body, dhcp.xid) != self.dhcp.xid) return; + // The reply must be about our hardware address, not a relayed one for someone else. + if (body[dhcp.hlen] != 6 or !std.mem.eql(u8, body[dhcp.chaddr..][0..6], &self.mac)) return; + + const kind_opt = dhcpOption(body, dhcp.opt_msg_type) orelse return; + if (kind_opt.len != 1) return; + self.counters.dhcp_rx += 1; + + switch (kind_opt[0]) { + dhcp.offer => { + if (self.dhcp.state != .selecting) return; + self.dhcp.offered = rdIp(body, dhcp.yiaddr); + if (std.mem.eql(u8, &self.dhcp.offered, &ip_any)) return; + // Option 54 is how the REQUEST names which offer it accepts. A server that omits + // it is out of spec; `siaddr` is the best fallback, and the sender is the last. + self.dhcp.server = dhcpOptionIp(body, dhcp.opt_server_id) orelse blk: { + const s = rdIp(body, dhcp.siaddr); + break :blk if (std.mem.eql(u8, &s, &ip_any)) src else s; + }; + self.dhcp.state = .requesting; + self.dhcpSend(dhcp.request); + self.dhcp.tries = 1; + self.dhcp.retry_ms = self.now_ms + dhcp_backoff_ms[0]; + }, + dhcp.ack => { + switch (self.dhcp.state) { + .requesting, .renewing, .rebinding => {}, + else => return, + } + const granted = rdIp(body, dhcp.yiaddr); + if (std.mem.eql(u8, &granted, &ip_any)) return; + self.dhcpBind(body, granted, src); + }, + dhcp.nak => { + switch (self.dhcp.state) { + .requesting, .renewing, .rebinding => {}, + else => return, + } + // RFC 2131 4.4.5: a NAK sends the client back to INIT. The lease is gone, so the + // address goes with it - continuing to use it would be squatting. + self.dhcpStart(); + }, + else => {}, + } + } + + fn dhcpBind(self: *Stack, body: []const u8, granted: Ip4, src: Ip4) void { + self.addr = granted; + self.mask = dhcpOptionIp(body, dhcp.opt_subnet_mask) orelse .{ 255, 255, 255, 0 }; + self.gw = dhcpOptionIp(body, dhcp.opt_router) orelse ip_any; + self.dns = dhcpOptionIp(body, dhcp.opt_dns); + if (dhcpOptionIp(body, dhcp.opt_server_id)) |s| self.dhcp.server = s else if (std.mem.eql(u8, &self.dhcp.server, &ip_any)) { + self.dhcp.server = src; + } + + // RFC 2131 3.3. A server that sends no lease time is out of spec; an hour is a safe + // assumption, being short enough that a wrong guess self-corrects. + const lease = dhcpOptionU32(body, dhcp.opt_lease_time) orelse 3600; + self.dhcp.lease_s = lease; + if (lease == 0xffff_ffff) { + // Infinite lease: never renew. + self.dhcp.t1_ms = std.math.maxInt(u64); + self.dhcp.t2_ms = std.math.maxInt(u64); + self.dhcp.expire_ms = std.math.maxInt(u64); + } else { + // The server may state T1 and T2 itself; otherwise lwIP's derivation, which is RFC + // 2131 4.4.5's: half the lease, and seven eighths of it + // (`core/ipv4/dhcp.c:757` and `:766`). + const t1 = dhcpOptionU32(body, dhcp.opt_t1) orelse lease / 2; + const t2 = dhcpOptionU32(body, dhcp.opt_t2) orelse (lease / 8) * 7; + const base = self.now_ms; + self.dhcp.t1_ms = base + @as(u64, @min(t1, lease)) * 1000; + self.dhcp.t2_ms = base + @as(u64, @min(t2, lease)) * 1000; + self.dhcp.expire_ms = base + @as(u64, lease) * 1000; + } + self.dhcp.state = .bound; + self.dhcp.tries = 0; + self.dhcp.retry_ms = 0; + self.announce(); + } + + fn dhcpTick(self: *Stack) void { + // T1 while bound: start renewing. RFC 2131 4.4.5 requires a fresh transaction id, and the + // first REQUEST goes out on this same tick rather than one backoff later - a state change + // that transmits nothing is how a lease quietly expires while the client thinks it is + // renewing. + if (self.dhcp.state == .bound and self.now_ms >= self.dhcp.t1_ms) { + self.dhcp.state = .renewing; + self.dhcp.xid = self.draw(); + self.dhcp.started_ms = self.now_ms; + self.dhcp.tries = 0; + self.dhcp.retry_ms = 0; + } + switch (self.dhcp.state) { + .off, .bound => return, + .selecting, .requesting, .renewing, .rebinding => {}, + } + if (self.dhcp.state == .renewing and self.now_ms >= self.dhcp.t2_ms) { + // T2: the granting server is not answering. Ask anyone. + self.dhcp.state = .rebinding; + self.dhcp.tries = 0; + self.dhcp.retry_ms = 0; + } + if ((self.dhcp.state == .renewing or self.dhcp.state == .rebinding) and + self.now_ms >= self.dhcp.expire_ms) + { + // The lease is over. Give up the address before asking again: keeping it would mean + // using an address the server may already have given away. + self.dhcpStart(); + return; + } + if (self.dhcp.retry_ms != 0 and self.now_ms < self.dhcp.retry_ms) return; + const kind: u8 = if (self.dhcp.state == .selecting) dhcp.discover else dhcp.request; + self.dhcpSend(kind); + const idx = @min(self.dhcp.tries, dhcp_backoff_ms.len - 1); + self.dhcp.retry_ms = self.now_ms + dhcp_backoff_ms[idx]; + if (self.dhcp.tries < 255) self.dhcp.tries += 1; + } + + // ================================================================================= DNS + // + // One question, QTYPE=A, QCLASS=IN, recursion desired, over the UDP above. No cache, no + // search list, no NS or SOA handling, no TCP fallback on a truncated answer: this resolves + // the one name a device that fetches one URL has to resolve, and says so with a named error + // when it cannot. + // + // The hard part of DNS parsing is not the header, it is that a name in a resource record may + // be a compression pointer into anywhere earlier in the message (RFC 1035 4.1.4). A parser + // that follows those without a bound hangs on a message that points at itself, and such a + // message costs an attacker two bytes. `dnsSkipName` is where that is dealt with. + + /// Encode a dotted name into RFC 1035 4.1.2 wire form: each label prefixed with its length, + /// terminated by the zero-length root label. Returns the encoded length. + /// + /// `out` must be at least `dns_qname_max`, which the length check below makes sufficient: a + /// name of `n` text bytes with no trailing dot encodes to exactly `n + 2`. + fn dnsEncodeName(name: []const u8, out: []u8) DnsError!usize { + assert(out.len >= dns_qname_max); + if (name.len > dns_name_max) return error.NameTooLong; + // A trailing dot is the root label written out, and `example.com.` names the same node as + // `example.com`. Everything after it - an empty final label - is not. + var rest = name; + if (rest.len != 0 and rest[rest.len - 1] == '.') rest = rest[0 .. rest.len - 1]; + if (rest.len == 0) return error.NameInvalid; + + var o: usize = 0; + var labels = std.mem.splitScalar(u8, rest, '.'); + while (labels.next()) |label| { + // An empty label inside a name (`a..b`, or a leading dot) is not a name. + if (label.len == 0 or label.len > dns.label_max) return error.NameInvalid; + out[o] = @intCast(label.len); + @memcpy(out[o + 1 ..][0..label.len], label); + o += 1 + label.len; + } + out[o] = 0; + return o + 1; + } + + /// Compare an encoded name against the question we asked, ASCII-case-insensitively. RFC 4343: + /// label comparison ignores case, and a resolver is entitled to answer `0X4200.CAFE` to a + /// question about `0x4200.cafe`. Length bytes are 0-63 and so are never touched by the fold. + fn dnsQNameEql(a: []const u8, b: []const u8) bool { + if (a.len != b.len) return false; + for (a, b) |x, y| if (std.ascii.toLower(x) != std.ascii.toLower(y)) return false; + return true; + } + + /// Step over the name at `start` and return the offset of the byte after it - which for a + /// name that ends in a compression pointer is two bytes after the pointer, *not* wherever the + /// pointer led. `null` means the name is unparseable and the message is to be rejected. + /// + /// **Why this terminates.** Two independent bounds, because one of them is not enough: + /// + /// * A pointer must point strictly backwards (`target < here`). That alone is the check + /// most implementations stop at, and it is *not* sufficient: after jumping back the walk + /// moves forward again over labels, so a pointer at offset 12 to offset 10 and a label at + /// 10 that is two bytes long lands back at 12, and the pair loops forever with every + /// individual jump going backwards. + /// * So the jumps themselves are counted, and `dns_max_jumps` of them ends the name. That + /// is the bound that actually holds: the loop below does at most `dns_max_jumps` jumps + /// and, between them, walks labels whose lengths are positive, so it visits at most + /// `dns_max_jumps * msg.len` bytes and stops. A legitimate answer uses one jump per name. + fn dnsSkipName(msg: []const u8, start: usize) ?usize { + var i = start; + var jumps: u8 = 0; + // The offset after the name in the *message*, fixed by the first pointer taken. + var after: ?usize = null; + while (true) { + if (i >= msg.len) return null; + const len = msg[i]; + if (len & dns.ptr_mask == dns.ptr_mask) { + if (i + 1 >= msg.len) return null; + const target = (@as(usize, len & 0x3f) << 8) | msg[i + 1]; + if (after == null) after = i + 2; + if (target >= i) return null; + jumps += 1; + if (jumps > dns_max_jumps) return null; + i = target; + continue; + } + // 0x40 and 0x80 are the reserved label types of RFC 1035 4.1.4 / RFC 6891; neither is + // something this stack can skip a known number of bytes past, so neither is accepted. + if (len & dns.ptr_mask != 0) return null; + if (len == 0) return after orelse i + 1; + i += 1 + @as(usize, len); + if (i > msg.len) return null; + } + } + + /// Build and send the query held in `self.query`. Called for the first transmission and for + /// every retransmission, from the same fields, so the two cannot drift apart. + fn dnsSend(self: *Stack) void { + const src = self.addr orelse return; + const server = self.dns orelse return; + const qn_len: usize = self.query.qname_len; + const msg_len = dns.hlen + qn_len + 4; + const msg = self.tx[eth_hlen + ip4.hlen + udp.hlen ..][0..msg_len]; + wr16(msg, dns.id, self.query.id); + // RD only. Not AD, not CD, not EDNS0: this asks a recursive resolver for one A record and + // has nothing to validate with. + wr16(msg, dns.flags, dns.flag_rd); + wr16(msg, dns.qdcount, 1); + wr16(msg, dns.ancount, 0); + wr16(msg, dns.nscount, 0); + wr16(msg, dns.arcount, 0); + @memcpy(msg[dns.hlen..][0..qn_len], self.query.qname[0..qn_len]); + wr16(msg, dns.hlen + qn_len, dns.type_a); + wr16(msg, dns.hlen + qn_len + 2, dns.class_in); + self.counters.dns_tx += 1; + _ = self.emitUdp(src, self.query.local_port, server, dns.port, msg_len); + } + + fn dnsFail(self: *Stack, e: DnsError) void { + self.query.phase = .failed; + self.query.err = e; + } + + /// A datagram to the port the outstanding query was sent from. Called from inside `onFrame`. + /// + /// Everything that does not match the query is *ignored*, not failed: on a real network the + /// port this query owns will collect late answers to previous queries, scans, and whatever + /// else is loose on the segment, and any of those failing the query would be a denial of + /// service that costs one packet. Only a response that matches the source, the id and the + /// question can decide the query - and then it decides it either way. + fn dnsInput(self: *Stack, src: Ip4, sport: u16, msg: []const u8) void { + const server = self.dns orelse return; + if (!std.mem.eql(u8, &src, &server)) return; + if (sport != dns.port) return; + if (msg.len < dns.hlen) return; + if (rd16(msg, dns.id) != self.query.id) return; + + const flags = rd16(msg, dns.flags); + if (flags & dns.flag_qr == 0) return; // a query, not a response + if (rd16(msg, dns.qdcount) != 1) return; + + // The question, echoed. A server that answers a different question - or an attacker who + // guessed the id and the port but not the name - is not answering this. + const qn = self.query.qname[0..self.query.qname_len]; + var off = dns.hlen + qn.len + 4; + if (msg.len < off) return; + if (!dnsQNameEql(msg[dns.hlen..][0..qn.len], qn)) return; + if (rd16(msg, dns.hlen + qn.len) != dns.type_a) return; + if (rd16(msg, dns.hlen + qn.len + 2) != dns.class_in) return; + + self.counters.dns_rx += 1; + + const rcode = flags & dns.rcode_mask; + if (rcode != 0) { + self.dnsFail(if (rcode == dns.rcode_name_error) error.NameNotFound else error.DnsRefused); + return; + } + + // Walk the answer section and take the first A record. Walking rather than reading the + // first record is what makes a CNAME chain work: `0x4200.cafe` may answer with the CNAME + // and the A together, in that order, and a resolver that reads answer[0] gets a name. + var left = rd16(msg, dns.ancount); + while (left != 0) : (left -= 1) { + off = dnsSkipName(msg, off) orelse { + self.dnsFail(error.DnsMalformed); + return; + }; + if (off + dns.rr_fixed > msg.len) { + self.dnsFail(error.DnsMalformed); + return; + } + const rtype = rd16(msg, off); + const rclass = rd16(msg, off + 2); + const rdlen: usize = rd16(msg, off + 8); + off += dns.rr_fixed; + if (off + rdlen > msg.len) { + self.dnsFail(error.DnsMalformed); + return; + } + if (rtype == dns.type_a and rclass == dns.class_in and rdlen == 4) { + self.query.result = rdIp(msg, off); + self.query.phase = .done; + return; + } + off += rdlen; + } + // A well-formed answer with no A record in it: NODATA, or a CNAME chain this stack will + // not chase a second query down. + self.dnsFail(error.NameNotFound); + } + + fn dnsTick(self: *Stack) void { + if (self.query.phase != .waiting) return; + if (self.now_ms < self.query.retry_ms) return; + if (self.query.tries >= dns_backoff_ms.len) { + self.dnsFail(error.TimedOut); + return; + } + self.query.retry_ms = self.now_ms + dns_backoff_ms[self.query.tries]; + self.query.tries += 1; + self.counters.dns_retx += 1; + self.dnsSend(); + } + + /// Resolve `name` to an IPv4 address. + /// + /// **The protocol is `httpGet`'s, deliberately.** There is no clock and no `std.Io` here, so + /// there is nothing for a blocking call to block on: the first call sends the query and + /// returns `error.WouldBlock`, and the caller drives `tick` and `onFrame` and calls again + /// with the same name until an address or a real error comes back. + /// + /// const addr = while (true) { + /// stack.tick(hal.systimer.millis()); + /// while (transport.next()) |frame| stack.onFrame(frame); + /// if (stack.resolve("0x4200.cafe")) |a| break a + /// else |e| if (e != error.WouldBlock) return e; + /// }; + /// + /// The wait is bounded whether or not the caller bounds it: `dns_backoff_ms` retransmits + /// three times over 7 s and then answers `error.TimedOut`. Nothing here waits forever, and + /// the retransmissions happen in `tick`, so a caller that ticks and polls rarely still gets + /// them on time. + /// + /// One query is outstanding at a time. A call naming something else while one is in flight is + /// `error.Busy`; a call naming something else after one has finished starts a new query, + /// which is what makes the loop above safe to write for two names in a row. + pub fn resolve(self: *Stack, name: []const u8) DnsError!Ip4 { + // Encoded first, and compared in encoded form: `0x4200.cafe`, `0x4200.cafe.` and + // `0X4200.CAFE` are one name, and a caller that spells it differently between two polls + // of the same loop must not get `error.Busy` for it. + var wire: [dns_qname_max]u8 = undefined; + const wire_len = try dnsEncodeName(name, &wire); + const same = self.query.qname_len == wire_len and + dnsQNameEql(self.query.qname[0..wire_len], wire[0..wire_len]); + + switch (self.query.phase) { + .idle => {}, + .waiting => { + if (!same) return error.Busy; + return error.WouldBlock; + }, + // A finished query for this name is collected and the slot released. A finished query + // for a different name falls through and is replaced. + .done => if (same) { + self.query.phase = .idle; + return self.query.result; + }, + .failed => if (same) { + self.query.phase = .idle; + return self.query.err; + }, + } + + if (self.addr == null) return error.NoAddress; + if (self.dns == null) return error.NoDnsServer; + + @memcpy(self.query.qname[0..wire_len], wire[0..wire_len]); + self.query.qname_len = @intCast(wire_len); + self.query.id = @truncate(self.draw()); + // RFC 6335's dynamic range, as `tcpConnect` uses. Redrawn per query so a late answer to + // the previous one cannot be mistaken for this one even if the id happens to repeat. + self.query.local_port = @intCast(49152 + self.draw() % (65535 - 49152 + 1)); + self.query.result = ip_any; + self.query.err = error.WouldBlock; + self.query.phase = .waiting; + self.query.tries = 1; + self.query.retry_ms = self.now_ms + dns_backoff_ms[0]; + self.dnsSend(); + return error.WouldBlock; + } + + // ================================================================================= TCP + // + // One connection, client side only, one unacknowledged segment at a time. The send side is a + // single static buffer holding the whole request, so "retransmission" is always "send from + // `snd_una` again" and there is no retransmission queue. The receive side has no reassembly + // buffer at all: a segment that is not the next one expected is answered with a duplicate ACK + // and dropped. Three of those is a fast-retransmit signal to any modern peer, so the common + // case of one lost segment costs a round trip rather than an RTO - but a reordered segment + // costs a retransmission that a reassembly buffer would have avoided. That is the price of not + // having one, and on a Wi-Fi link where reordering is rare it is the right price. + + fn emitTcp(self: *Stack, flags: u8, seq: u32, data: []const u8, mss_opt: bool) bool { + const src = self.addr orelse return false; + const opt_len: usize = if (mss_opt) 4 else 0; + const seg_len = tcp.hlen + opt_len + data.len; + assert(seg_len <= mtu - ip4.hlen); + const seg = self.tx[eth_hlen + ip4.hlen ..][0..seg_len]; + + wr16(seg, tcp.src_port, self.tcp.local_port); + wr16(seg, tcp.dst_port, self.tcp.peer_port); + wr32(seg, tcp.seq, seq); + wr32(seg, tcp.ack, self.tcp.rcv_nxt); + const words: u16 = @intCast((tcp.hlen + opt_len) / 4); + wr16(seg, tcp.hdrlen_flags, (words << 12) | flags); + wr16(seg, tcp.window, self.rcvWindow()); + wr16(seg, tcp.chksum, 0); + wr16(seg, tcp.urgent, 0); + if (mss_opt) { + seg[tcp.hlen] = tcp.opt_mss; + seg[tcp.hlen + 1] = 4; + wr16(seg, tcp.hlen + 2, tcp_mss); + } + if (data.len != 0) @memcpy(seg[tcp.hlen + opt_len ..], data); + wr16(seg, tcp.chksum, transportChecksum(src, self.tcp.peer_ip, .tcp, seg)); + + self.counters.tcp_tx += 1; + return self.emitIp(src, self.tcp.peer_ip, .tcp, seg_len); + } + + /// The window to advertise: real back-pressure, not a constant. Everything accepted is consumed + /// synchronously into the HTTP head buffer or the caller's `out`, so the window is whatever + /// room is left there, capped at one MSS. Advertising a fixed window while the consumer had no + /// room left would turn "the response is bigger than your buffer" into a silently dropped + /// segment and an RTO storm. + fn rcvWindow(self: *Stack) u16 { + const room: usize = switch (self.http.phase) { + .head => (http_head_max - self.http.head_len) + self.http.out.len, + // Chunked framing - the CRLF closing each chunk, the zero-length chunk, the trailer + // section and the final CRLF - is consumed and discarded rather than delivered, so it + // needs window that `out` does not account for. Without this a body that exactly + // fills `out` closes the window before its own terminator can arrive, and the request + // stalls until the RTO gives up on a peer that is behaving perfectly. + .body => (self.http.out.len - self.http.out_len) + + @as(usize, if (self.http.chunked) http_framing_max + 16 else 0), + else => tcp_window, + }; + return @intCast(@min(room, tcp_window)); + } + + /// Reset the connection and fail the request. RST is sent unless the peer sent one. + fn tcpAbort(self: *Stack, err: HttpError, send_rst: bool) void { + if (send_rst and self.tcp.state != .closed and self.tcp.state != .arp_wait) { + _ = self.emitTcp(tcp.rst | tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + } + self.tcp.state = .closed; + if (self.http.phase == .head or self.http.phase == .body) { + self.http.phase = .failed; + self.http.err = err; + } + } + + /// Set up the connection block for a fresh connect. Written field by field on purpose: the + /// obvious `self.tcp = .{ ... }` would assign `tx` from the struct's `undefined` default, which + /// in a safe build overwrites the request bytes with 0xAA, and in a release build memsets half + /// a kilobyte for nothing. + fn tcpConnect(self: *Stack, peer: Ip4, port: u16) void { + // A fresh ephemeral port every time. RFC 6335's dynamic range is 49152-65535, and moving + // through it is what makes the short TIME_WAIT above safe. + const span: u32 = 65535 - 49152 + 1; + const iss = self.draw(); + self.tcp.state = .arp_wait; + self.tcp.peer_ip = peer; + self.tcp.peer_port = port; + self.tcp.local_port = @intCast(49152 + self.draw() % span); + self.tcp.iss = iss; + self.tcp.snd_una = iss; + self.tcp.snd_nxt = iss; + self.tcp.snd_wnd = 0; + self.tcp.snd_mss = tcp_default_mss; + self.tcp.fin_queued = false; + self.tcp.peer_fin = false; + self.tcp.rcv_nxt = 0; + self.tcp.rto_deadline_ms = 0; + self.tcp.rto_ms = tcp_rto_initial_ms; + self.tcp.retries = 0; + self.tcp.close_deadline_ms = 0; + self.tcp.tx_len = 0; + self.arp_tries = 0; + self.arp_retry_ms = 0; + } + + fn tcpSendSyn(self: *Stack) void { + self.tcp.state = .syn_sent; + self.tcp.snd_nxt = self.tcp.iss +% 1; + _ = self.emitTcp(tcp.syn, self.tcp.iss, &.{}, true); + self.armRto(); + } + + fn armRto(self: *Stack) void { + self.tcp.rto_deadline_ms = self.now_ms + self.tcp.rto_ms; + } + + /// Send as much of the request as the peer's window and MSS allow, then the FIN if the whole + /// request has gone out. One segment in flight, so this sends at most one segment per call. + /// + /// Only `established` sends: receiving the peer's FIN does not move the state, it sets + /// `peer_fin`, so this stays the single place that decides what goes on the wire and the state + /// only ever changes when a FIN of ours actually leaves. + fn tcpSendData(self: *Stack) void { + if (self.tcp.state != .established) return; + // Nothing outstanding is the precondition for sending: this is the fixed window of one. + if (seqLt(self.tcp.snd_una, self.tcp.snd_nxt)) return; + + const end = self.tcp.dataEnd(); + if (seqLt(self.tcp.snd_nxt, end)) { + const off: usize = self.tcp.snd_nxt -% self.tcp.dataStart(); + const remaining = self.tcp.tx_len - off; + const window: usize = self.tcp.snd_una +% self.tcp.snd_wnd -% self.tcp.snd_nxt; + const n = @min(@min(remaining, self.tcp.snd_mss), @max(window, 1)); + // PSH on the last segment of the request: the peer's application should see it without + // waiting for more. RFC 793 has no requirement here; every HTTP server expects it. + const last = off + n == self.tcp.tx_len; + const flags: u8 = tcp.ack_f | (if (last) tcp.psh else 0); + const seq = self.tcp.snd_nxt; + self.tcp.snd_nxt = seq +% @as(u32, @intCast(n)); + _ = self.emitTcp(flags, seq, self.tcp.tx[off..][0..n], false); + self.armRto(); + return; + } + if (self.tcp.fin_queued and self.tcp.snd_nxt == end) { + const seq = self.tcp.snd_nxt; + self.tcp.snd_nxt = seq +% 1; + _ = self.emitTcp(tcp.fin | tcp.ack_f, seq, &.{}, false); + self.armRto(); + // RFC 793's FIN-WAIT-1 if we closed first, its CLOSING/LAST-ACK if the peer did. Both + // of the latter are `last_ack` here: they differ only in which ACK is still owed, and + // `closeCheck` settles that from the sequence numbers. + self.tcp.state = if (self.tcp.peer_fin) .last_ack else .fin_wait_1; + } + } + + /// Half-close: everything we mean to send has been sent, so send FIN once the data is out. + fn tcpFinish(self: *Stack) void { + if (self.tcp.fin_queued) return; + self.tcp.fin_queued = true; + self.tcpSendData(); + } + + /// Both directions closed and our FIN acknowledged: nothing is left in flight, so the + /// connection block can be released after TIME_WAIT. Called once at the end of every segment, + /// which covers both orders of arrival - the peer's FIN then its ACK, or the reverse. + fn closeCheck(self: *Stack) void { + switch (self.tcp.state) { + .fin_wait_1, .fin_wait_2, .last_ack => {}, + else => return, + } + if (!self.tcp.peer_fin) return; + if (self.tcp.snd_una != self.tcp.snd_nxt) return; + self.tcp.state = .time_wait; + self.tcp.rto_deadline_ms = 0; + self.tcp.close_deadline_ms = self.now_ms + tcp_time_wait_ms; + } + + fn tcpInput(self: *Stack, src: Ip4, dst: Ip4, seg: []const u8) void { + if (seg.len < tcp.hlen) { + self.counters.rx_dropped += 1; + return; + } + const hf = rd16(seg, tcp.hdrlen_flags); + const hlen = @as(usize, hf >> 12) * 4; + if (hlen < tcp.hlen or hlen > seg.len) { + self.counters.rx_dropped += 1; + return; + } + if (!transportChecksumOk(src, dst, .tcp, seg, rd16(seg, tcp.chksum))) { + self.counters.checksum_bad += 1; + return; + } + const flags: u8 = @truncate(hf & 0x3f); + const sport = rd16(seg, tcp.src_port); + const dport = rd16(seg, tcp.dst_port); + + if (self.tcp.state == .closed or + dport != self.tcp.local_port or + sport != self.tcp.peer_port or + !std.mem.eql(u8, &src, &self.tcp.peer_ip)) + { + // Not for our one connection. A real stack would RST; a client with no listening port + // gains nothing by telling a scanner it is there. + self.counters.rx_dropped += 1; + return; + } + self.counters.tcp_rx += 1; + + const seq = rd32(seg, tcp.seq); + const ackno = rd32(seg, tcp.ack); + const data = seg[hlen..]; + + if (flags & tcp.rst != 0) { + self.counters.tcp_rst_rx += 1; + // RFC 5961 3: only a RST whose sequence number is the next one expected may tear the + // connection down. Anything else gets a challenge ACK, which is also what stops a + // blind off-path reset. + if (self.tcp.state == .syn_sent) { + // In SYN-SENT the RST is validated by its ACK instead: there is no rcv_nxt yet. + if (flags & tcp.ack_f != 0 and ackno == self.tcp.snd_nxt) self.tcpAbort(error.ConnectionReset, false); + return; + } + if (seq == self.tcp.rcv_nxt) { + self.tcpAbort(error.ConnectionReset, false); + } else { + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + } + return; + } + + if (self.tcp.state == .syn_sent) { + if (flags & tcp.syn == 0) { + self.counters.rx_dropped += 1; + return; + } + if (flags & tcp.ack_f == 0) { + // A simultaneous open. Nothing on the other end of an HTTP GET does this. + self.counters.rx_dropped += 1; + return; + } + if (ackno != self.tcp.iss +% 1) { + // Not acknowledging our SYN: an old duplicate. RFC 793 says reset it. + _ = self.emitTcp(tcp.rst, ackno, &.{}, false); + return; + } + self.tcp.rcv_nxt = seq +% 1; + self.tcp.snd_una = ackno; + self.tcp.snd_wnd = rd16(seg, tcp.window); + self.tcp.snd_mss = parseMss(seg[tcp.hlen..hlen]) orelse tcp_default_mss; + self.tcp.state = .established; + self.tcp.rto_ms = tcp_rto_initial_ms; + self.tcp.retries = 0; + // The ACK completing the handshake carries the first data segment, which is one frame + // saved and what every other stack does. + self.tcp.rto_deadline_ms = 0; + self.tcpSendData(); + if (self.tcp.snd_nxt == self.tcp.snd_una) { + // Nothing to send yet; the handshake still needs acknowledging. + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + } + return; + } + + // A duplicate SYN in an established connection is either a retransmitted SYN whose ACK was + // lost - answer with an ACK - or an attack. Never a reason to re-open. + if (flags & tcp.syn != 0 and seqLt(seq, self.tcp.rcv_nxt)) { + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + return; + } + + if (flags & tcp.ack_f != 0) self.tcpAck(ackno, rd16(seg, tcp.window)); + + // ---- receive side + var payload = data; + var accept = false; + if (payload.len != 0) { + if (seqLe(seq, self.tcp.rcv_nxt) and seqGt(seq +% @as(u32, @intCast(payload.len)), self.tcp.rcv_nxt)) { + // Overlaps what we already have: trim the duplicate prefix. A retransmission after + // a lost ACK arrives exactly like this, and rejecting it would deadlock. + const skip: usize = self.tcp.rcv_nxt -% seq; + payload = payload[skip..]; + accept = true; + } else if (seqLe(seq +% @as(u32, @intCast(payload.len)), self.tcp.rcv_nxt)) { + // Wholly old. Re-acknowledge so the peer stops. + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + return; + } else { + // Out of order, and there is nowhere to keep it. The duplicate ACK below is the + // signal that makes the peer resend. + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + return; + } + } + + if (accept) { + // Never accept more than the window we advertised. + const room = self.rcvWindow(); + if (payload.len > room) payload = payload[0..room]; + self.tcp.rcv_nxt +%= @intCast(payload.len); + // Consuming the data may itself put a segment on the wire - completing the body sends + // our FIN - and every segment carries `rcv_nxt`, so a separate ACK would be a wasted + // frame. Counting is the honest way to know: anything emitted has already acknowledged + // this data, and nothing emitted means we still owe an ACK. + const tx_before = self.counters.tcp_tx; + self.httpOnData(payload); + if (self.tcp.state == .closed) return; // httpOnData failed and aborted + if (self.counters.tcp_tx == tx_before) { + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + } + } + + // ---- FIN, in order only. An out-of-order FIN names a sequence number beyond data we have + // not seen, and honouring it would close the connection over a hole. + if (flags & tcp.fin != 0) { + const fin_seq = seq +% @as(u32, @intCast(data.len)); + const in_order = fin_seq == self.tcp.rcv_nxt; + // A FIN we have already consumed, arriving again because our ACK was lost. It must be + // re-acknowledged or the peer retransmits until it gives up and resets. + const duplicate = self.tcp.peer_fin and fin_seq +% 1 == self.tcp.rcv_nxt; + if (in_order and !self.tcp.peer_fin) { + self.tcp.rcv_nxt +%= 1; + self.tcp.peer_fin = true; + self.httpOnEof(); + } + if (in_order or duplicate) { + // Our own FIN, if it has not gone yet, acknowledges the peer's on the way out. + const tx_before = self.counters.tcp_tx; + self.tcpFinish(); + if (self.counters.tcp_tx == tx_before) { + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + } + } + } + + self.closeCheck(); + } + + fn tcpAck(self: *Stack, ackno: u32, window: u16) void { + // An ACK ahead of what we sent is invalid; an old one is a duplicate. + if (seqGt(ackno, self.tcp.snd_nxt)) return; + self.tcp.snd_wnd = window; + if (seqLe(ackno, self.tcp.snd_una)) { + // Duplicate ACK. With one segment in flight there is nothing to fast-retransmit. + return; + } + self.tcp.snd_una = ackno; + self.tcp.retries = 0; + self.tcp.rto_ms = tcp_rto_initial_ms; + if (self.tcp.snd_una == self.tcp.snd_nxt) { + self.tcp.rto_deadline_ms = 0; // nothing outstanding + } else { + self.armRto(); + } + if (self.tcp.state == .fin_wait_1 and self.tcp.snd_una == self.tcp.snd_nxt) { + self.tcp.state = .fin_wait_2; + self.tcp.close_deadline_ms = self.now_ms + tcp_fin_wait2_ms; + } + // Window opened or data acknowledged: there may be more to send. + self.tcpSendData(); + } + + /// RFC 793 3.1 option format: kind, then for kinds above 1 a length byte covering both. + fn parseMss(opts: []const u8) ?u16 { + var i: usize = 0; + while (i < opts.len) { + const kind = opts[i]; + if (kind == tcp.opt_end) return null; + if (kind == tcp.opt_nop) { + i += 1; + continue; + } + if (i + 2 > opts.len) return null; + const len = opts[i + 1]; + if (len < 2 or i + len > opts.len) return null; + if (kind == tcp.opt_mss and len == 4) { + const v = rd16(opts, i + 2); + // Below RFC 1122's floor a peer's MSS is not believable; above our MTU it cannot + // be honoured anyway. + return @min(@max(v, 64), tcp_mss); + } + i += len; + } + return null; + } + + fn tcpTick(self: *Stack) void { + switch (self.tcp.state) { + .closed => {}, + .arp_wait => { + if (self.arpLookup(self.tcpNextHop())) |_| { + self.tcpSendSyn(); + return; + } + if (self.arp_retry_ms != 0 and self.now_ms < self.arp_retry_ms) return; + if (self.arp_tries >= arp_max_tries) { + self.tcpAbort(error.HostUnreachable, false); + return; + } + self.arpRequest(self.tcpNextHop()); + self.arp_tries += 1; + self.arp_retry_ms = self.now_ms + arp_retry_ms; + }, + .fin_wait_2 => { + // Our FIN is acknowledged and nothing is outstanding, so there is no RTO to run: + // the only thing left is the peer's FIN, and this is how long we wait for it. + if (self.now_ms >= self.tcp.close_deadline_ms) self.tcp.state = .closed; + }, + .time_wait => { + if (self.now_ms >= self.tcp.close_deadline_ms) self.tcp.state = .closed; + }, + else => { + if (self.tcp.rto_deadline_ms == 0) return; + if (self.now_ms < self.tcp.rto_deadline_ms) return; + if (self.tcp.retries >= tcp_max_retries) { + self.tcpAbort(error.TimedOut, true); + return; + } + self.tcp.retries += 1; + self.counters.tcp_retx += 1; + // Exponential backoff, RFC 6298 5.5. + self.tcp.rto_ms = @min(self.tcp.rto_ms * 2, tcp_rto_max_ms); + self.tcpRetransmit(); + }, + } + } + + fn tcpNextHop(self: *Stack) Ip4 { + return if (self.onLink(self.tcp.peer_ip)) self.tcp.peer_ip else self.gw; + } + + /// Go back to `snd_una` and send again. With one segment in flight this is the whole of + /// retransmission: there is no queue to walk and no partial-ACK case to handle. + fn tcpRetransmit(self: *Stack) void { + const una = self.tcp.snd_una; + if (una == self.tcp.iss) { + // The SYN. Its MSS option must be repeated: a peer that only ever sees the + // retransmission would otherwise assume 536. + self.tcp.snd_nxt = self.tcp.iss; + self.tcpSendSyn(); + return; + } + const end = self.tcp.dataEnd(); + if (seqLt(una, end)) { + self.tcp.snd_nxt = una; + self.tcpSendData(); + return; + } + if (self.tcp.fin_queued and una == end) { + self.tcp.snd_nxt = una; + // `tcpSendData` re-sends the FIN and re-arms, but it refuses to run in FIN_WAIT_1 + // (which is where a lost FIN leaves us), so the segment is emitted directly. + _ = self.emitTcp(tcp.fin | tcp.ack_f, una, &.{}, false); + self.tcp.snd_nxt = una +% 1; + self.armRto(); + return; + } + // Nothing identifiable outstanding: a bare ACK, which costs one frame and cannot hurt. + _ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false); + self.armRto(); + } + + // ================================================================================ HTTP + + /// Fetch `path` from `host:port` over HTTP/1.1 and write the body to `out`, sending the + /// address literal as the `Host:` header. Exactly `httpGetHost(host, null, ...)`; see there + /// for the protocol, which is the whole of how this is used. + pub fn httpGet(self: *Stack, host: [4]u8, port: u16, path: []const u8, out: []u8) HttpError!usize { + return self.httpGetHost(host, null, port, path, out); + } + + /// Fetch `path` from `host:port` over HTTP/1.1 and write the body to `out`. + /// + /// `name` is the `Host:` header. `null` sends the address literal - `Host: 192.168.1.90` - + /// which is right for a bare address and is what `httpGet` does. A name is what a + /// name-based virtual host requires: one address behind a CDN serves thousands of sites and + /// picks between them on this header alone, so `Host: 104.21.46.8` gets the CDN's own error + /// page and never the site. The address is still where the connection goes; the name only + /// ever appears in the header, and nothing here resolves it - `resolve` does that, and the + /// two are separate because a caller may have the address already. + /// + /// The port is appended as `:port` only when it is not 80, name or no name. RFC 7230 5.4. + /// + /// **This does not block, and it is not a one-shot call.** There is no `std.Io` here and no + /// clock, so there is nothing for a blocking call to block on: the frames that carry the + /// response arrive through `onFrame` and time advances through `tick`, both of which are the + /// caller's to drive. So the first call starts the request and returns `error.WouldBlock`, and + /// the caller keeps driving and keeps calling with the same arguments until it returns a length: + /// + /// while (true) { + /// stack.tick(hal.systimer.millis()); + /// while (transport.next()) |frame| stack.onFrame(frame); + /// if (stack.httpGetHost(addr, "0x4200.cafe", 80, "/", &buf)) |n| break :done buf[0..n] + /// else |e| if (e != error.WouldBlock) return e; + /// } + /// + /// `out` is borrowed until the request completes: it is written to from inside `onFrame` as the + /// body arrives, so it must not move or be reused meanwhile. Calling with different arguments + /// while a request is in flight returns `error.Busy` rather than quietly abandoning the first, + /// and `name` is one of those arguments: two requests to one address for one path but + /// different virtual hosts are different requests. + /// + /// `Content-Length` is honoured, and so is `Transfer-Encoding: chunked` - the body handed back + /// is decoded, with no framing bytes in it. A response with neither ends at the peer's FIN, + /// which is why the request says `Connection: close`. + pub fn httpGetHost( + self: *Stack, + host: [4]u8, + name: ?[]const u8, + port: u16, + path: []const u8, + out: []u8, + ) HttpError!usize { + // The name is folded into the path hash rather than given a field of its own: `Stack` has + // a 4 KiB budget, and what this has to distinguish is "the same call again" from "a + // different call", which a hash does exactly. Seeding with the name's hash rather than + // concatenating keeps `null` (seed 0) distinct from any name, including the empty one. + const req_hash = std.hash.Wyhash.hash( + if (name) |nm| std.hash.Wyhash.hash(0x486f_7374, nm) else 0, + path, + ); + switch (self.http.phase) { + .idle => {}, + .head, .body => { + if (!std.mem.eql(u8, &self.http.req_host, &host) or + self.http.req_port != port or + self.http.req_hash != req_hash or + self.http.out.ptr != out.ptr or + self.http.out.len != out.len) return error.Busy; + return error.WouldBlock; + }, + .complete => { + const n = self.http.out_len; + self.http.phase = .idle; + return n; + }, + .failed => { + const e = self.http.err; + self.http.phase = .idle; + return e; + }, + } + + if (self.addr == null) return error.NoAddress; + + // The request, built once into the TCP send buffer where it stays until acknowledged. + var w: RequestWriter = .{ .buf = &self.tcp.tx }; + w.str("GET "); + w.str(if (path.len == 0) "/" else path); + w.str(" HTTP/1.1\r\nHost: "); + if (name) |nm| w.str(nm) else w.ipv4(host); + if (port != 80) { + w.str(":"); + w.dec(port); + } + // Connection: close is not politeness, it is the framing: it is what makes a response with + // no Content-Length terminable, and it is what makes the peer's FIN the end of the body. + w.str("\r\nUser-Agent: zig-p4/0.1\r\nAccept: */*\r\nConnection: close\r\n\r\n"); + if (w.overflow) return error.RequestTooLong; + + self.http = .{ + .phase = .head, + .out = out, + .req_host = host, + .req_port = port, + .req_hash = req_hash, + }; + self.tcpConnect(host, port); + self.tcp.tx_len = w.i; + self.tcp.fin_queued = false; + // A MAC address may already be known, in which case the SYN goes out now rather than one + // `tick` later. + if (self.arpLookup(self.tcpNextHop()) != null) { + self.tcpSendSyn(); + } else { + self.arpRequest(self.tcpNextHop()); + self.arp_tries = 1; + self.arp_retry_ms = self.now_ms + arp_retry_ms; + } + return error.WouldBlock; + } + + /// A bounds-checked append into a fixed buffer. Overflow is recorded, not asserted: a caller's + /// long path is a request error, not a bug in this file. + const RequestWriter = struct { + buf: []u8, + i: usize = 0, + overflow: bool = false, + + fn str(self: *RequestWriter, s: []const u8) void { + if (self.overflow or self.i + s.len > self.buf.len) { + self.overflow = true; + return; + } + @memcpy(self.buf[self.i..][0..s.len], s); + self.i += s.len; + } + fn dec(self: *RequestWriter, v: u32) void { + var tmp: [10]u8 = undefined; + var n: usize = 0; + var x = v; + while (true) { + tmp[n] = '0' + @as(u8, @intCast(x % 10)); + n += 1; + x /= 10; + if (x == 0) break; + } + while (n > 0) { + n -= 1; + self.str(tmp[n .. n + 1]); + } + } + fn ipv4(self: *RequestWriter, a: Ip4) void { + for (a, 0..) |b, k| { + if (k != 0) self.str("."); + self.dec(b); + } + } + }; + + /// Fail the request and reset the connection. The phase is set before `tcpAbort`, which would + /// otherwise overwrite `err` with its own argument on the way past. + fn httpFail(self: *Stack, e: HttpError) void { + self.http.phase = .failed; + self.http.err = e; + self.tcpAbort(e, true); + } + + /// In-order TCP payload. Called from inside `onFrame`. + fn httpOnData(self: *Stack, bytes: []const u8) void { + var rest = bytes; + if (self.http.phase == .head) { + const room = http_head_max - self.http.head_len; + const n = @min(room, rest.len); + @memcpy(self.http.head[self.http.head_len..][0..n], rest[0..n]); + const scan_from = self.http.head_len -| 3; + self.http.head_len += n; + rest = rest[n..]; + + const blank = std.mem.indexOfPos(u8, self.http.head[0..self.http.head_len], scan_from, "\r\n\r\n") orelse { + if (self.http.head_len == http_head_max) self.httpFail(error.HttpHeadersTooLong); + return; + }; + const head_end = blank + 4; + // Anything the head buffer swallowed past the blank line is body. This is the case a + // test has to cover deliberately, because it only happens when a segment boundary does + // not coincide with the end of the headers - which on a real server is most of the time. + const spill = self.http.head[head_end..self.http.head_len]; + self.parseHead(self.http.head[0..blank]) catch |e| { + self.httpFail(e); + return; + }; + self.http.phase = .body; + // `spill` aliases `self.http.head`, and `httpBody` only ever writes to `self.http.out`, + // so passing it through is safe. Copy first if that ever stops being true. + // + // It is called unconditionally, even when `spill` is empty: that is what completes a + // `Content-Length: 0` response, whose body is over the moment its headers are. + self.httpBody(spill); + if (self.http.phase != .body) return; + } + if (rest.len != 0) self.httpBody(rest); + } + + /// Status line and headers, without the terminating blank line. + fn parseHead(self: *Stack, head: []const u8) HttpError!void { + var lines = std.mem.splitSequence(u8, head, "\r\n"); + const status_line = lines.next() orelse return error.HttpMalformed; + // "HTTP/1.1 200 OK": version, space, three digits. + if (status_line.len < 12) return error.HttpMalformed; + if (!std.mem.startsWith(u8, status_line, "HTTP/1.")) return error.HttpMalformed; + if (status_line[8] != ' ') return error.HttpMalformed; + var code: u16 = 0; + for (status_line[9..12]) |c| { + if (c < '0' or c > '9') return error.HttpMalformed; + code = code * 10 + (c - '0'); + } + self.http.status = code; + self.http.content_length = null; + self.http.chunked = false; + + while (lines.next()) |line| { + if (line.len == 0) continue; + const colon = std.mem.indexOfScalar(u8, line, ':') orelse continue; + const name = line[0..colon]; + const value = std.mem.trim(u8, line[colon + 1 ..], " \t"); + // RFC 7230 3.2: field names are case-insensitive. Servers vary, and a stack that + // compares them exactly works against nginx and fails against something else. + if (std.ascii.eqlIgnoreCase(name, "content-length")) { + self.http.content_length = std.fmt.parseInt(usize, value, 10) catch + return error.HttpMalformed; + } else if (std.ascii.eqlIgnoreCase(name, "transfer-encoding")) { + // RFC 7230 3.3.1: the final coding decides the framing. Exactly two are + // understood - `chunked`, which frames the body, and `identity`, which does not - + // and a list, or a coding that transforms the bytes, is refused. Guessing at + // `gzip` would hand the caller compressed data and call it a body. + if (std.ascii.eqlIgnoreCase(value, "chunked")) { + self.http.chunked = true; + } else if (!std.ascii.eqlIgnoreCase(value, "identity")) { + return error.UnsupportedTransferEncoding; + } + } + } + if (self.http.chunked) { + // RFC 7230 3.3.3 case 3: when both are present the chunked framing wins and + // `Content-Length` must be ignored - it is the classic request-smuggling + // disagreement, and a response that carries both is not to be believed twice. + self.http.content_length = null; + self.http.chunk = .size; + self.http.chunk_left = 0; + self.http.chunk_digit = false; + self.http.chunk_skip = 0; + } + // A response whose body cannot possibly fit is refused now rather than after copying most + // of it: the caller gets a clean error instead of a truncated buffer. A chunked response + // announces no total, so its equivalent check is per chunk, in `httpChunkedBody`. + if (self.http.content_length) |len| { + if (len > self.http.out.len) return error.StreamTooLong; + } + } + + fn httpBody(self: *Stack, bytes: []const u8) void { + if (self.http.chunked) return self.httpChunkedBody(bytes); + var b = bytes; + if (self.http.content_length) |len| { + const want = len - self.http.out_len; + if (b.len > want) b = b[0..want]; + } + if (self.http.out_len + b.len > self.http.out.len) { + self.httpFail(error.StreamTooLong); + return; + } + @memcpy(self.http.out[self.http.out_len..][0..b.len], b); + self.http.out_len += b.len; + if (self.http.content_length) |len| { + if (self.http.out_len >= len) self.httpComplete(); + } + } + + /// Charge `n` bytes against the framing budget. False means the request has been failed and + /// the decoder must stop. + fn chunkSkip(self: *Stack, n: usize) bool { + const total = @as(usize, self.http.chunk_skip) + n; + if (total > http_framing_max) { + self.httpFail(error.HttpHeadersTooLong); + return false; + } + self.http.chunk_skip = @intCast(total); + return true; + } + + /// RFC 7230 4.1 chunked decoding, resumable between any two bytes. + /// + /// The decoder's whole position lives in `http.chunk`, `chunk_left`, `chunk_digit` and + /// `chunk_skip`, and `bytes` is whatever the last segment happened to carry. Nothing is + /// buffered and nothing is looked ahead at: a size split across two segments accumulates a + /// digit at a time, a CRLF split across two segments is two states, and a chunk's data is + /// copied out as it arrives however it is cut up. That is not a hypothetical - a 1,460-byte + /// segment ends where the server's writes ended, which is nowhere in particular. + /// + /// `out` receives decoded data only. No size, no extension, no CRLF and no trailer byte is + /// ever copied into it, and every failure is a named error rather than a short body. + fn httpChunkedBody(self: *Stack, bytes: []const u8) void { + var b = bytes; + while (b.len != 0) { + switch (self.http.chunk) { + .size => { + const c = b[0]; + const digit: ?u8 = switch (c) { + '0'...'9' => c - '0', + 'a'...'f' => c - 'a' + 10, + 'A'...'F' => c - 'A' + 10, + else => null, + }; + b = b[1..]; + if (digit) |d| { + // Checked, not truncated: a size that does not fit `usize` is a malformed + // message, and wrapping it would turn a hostile header into a short read + // that looks like a complete body. + if (self.http.chunk_left > (std.math.maxInt(usize) - @as(usize, d)) / 16) { + self.httpFail(error.HttpChunkMalformed); + return; + } + self.http.chunk_left = self.http.chunk_left * 16 + d; + self.http.chunk_digit = true; + continue; + } + // RFC 7230 4.1 is `1*HEXDIG`. Without this an empty line reads as a chunk of + // size zero, which is the terminator, which ends the body early. + if (!self.http.chunk_digit) { + self.httpFail(error.HttpChunkMalformed); + return; + } + self.http.chunk_skip = 0; + switch (c) { + ';' => self.http.chunk = .ext, + '\r' => self.http.chunk = .size_lf, + else => { + self.httpFail(error.HttpChunkMalformed); + return; + }, + } + }, + .ext => { + // chunk-ext is skipped whole: nothing here depends on one, so the only thing + // that matters is finding the CR that ends the header - possibly not in this + // segment at all. + const cr = std.mem.indexOfScalar(u8, b, '\r'); + const n = cr orelse b.len; + if (!self.chunkSkip(n)) return; + b = b[n..]; + if (cr != null) { + b = b[1..]; + self.http.chunk = .size_lf; + } + }, + .size_lf => { + if (b[0] != '\n') { + self.httpFail(error.HttpChunkMalformed); + return; + } + b = b[1..]; + if (self.http.chunk_left == 0) { + // The zero-length chunk. What follows is the trailer section, and the + // body is not complete until its final CRLF. + self.http.chunk_skip = 0; + self.http.chunk = .trailer; + } else { + // Refused on the header rather than part-way through the copy, which is + // what `Content-Length` gets: the caller sees the error before the buffer + // has been half filled with a body it will never be given. + if (self.http.chunk_left > self.http.out.len - self.http.out_len) { + self.httpFail(error.StreamTooLong); + return; + } + self.http.chunk = .data; + } + }, + .data => { + // In bounds by construction: `.size_lf` refused any chunk larger than the room + // left, and this only ever takes `chunk_left` of it. + const n = @min(self.http.chunk_left, b.len); + @memcpy(self.http.out[self.http.out_len..][0..n], b[0..n]); + self.http.out_len += n; + self.http.chunk_left -= n; + b = b[n..]; + if (self.http.chunk_left == 0) self.http.chunk = .data_cr; + }, + .data_cr => { + if (b[0] != '\r') { + self.httpFail(error.HttpChunkMalformed); + return; + } + b = b[1..]; + self.http.chunk = .data_lf; + }, + .data_lf => { + if (b[0] != '\n') { + self.httpFail(error.HttpChunkMalformed); + return; + } + b = b[1..]; + // `.data` is only ever left with the chunk exhausted, so the accumulator the + // next size builds in already reads zero and is not re-zeroed here. Asserted + // rather than assumed: re-zeroing would be dead code that hides the day the + // invariant stops holding, and a stale count would be silent. + assert(self.http.chunk_left == 0); + self.http.chunk_digit = false; + self.http.chunk = .size; + }, + .trailer => { + if (!self.chunkSkip(1)) return; + const cr = b[0] == '\r'; + b = b[1..]; + self.http.chunk = if (cr) .end_lf else .trailer_line; + }, + .trailer_line => { + const cr = std.mem.indexOfScalar(u8, b, '\r'); + const n = cr orelse b.len; + if (!self.chunkSkip(n)) return; + b = b[n..]; + if (cr != null) { + b = b[1..]; + self.http.chunk = .trailer_lf; + } + }, + .trailer_lf => { + if (b[0] != '\n') { + self.httpFail(error.HttpChunkMalformed); + return; + } + b = b[1..]; + self.http.chunk = .trailer; + }, + .end_lf => { + if (b[0] != '\n') { + self.httpFail(error.HttpChunkMalformed); + return; + } + self.httpComplete(); + // Anything after the final CRLF belongs to a response this connection will + // never ask for: `Connection: close` was sent, and the FIN follows. + return; + }, + } + } + } + + fn httpComplete(self: *Stack) void { + self.http.phase = .complete; + // The body is in hand; close our half. Reading further would only cost frames. + self.tcpFinish(); + } + + /// The peer closed. Whether that completes the response depends on the framing. + fn httpOnEof(self: *Stack) void { + switch (self.http.phase) { + .body => { + if (self.http.chunked) { + // The zero-length chunk and its trailer never arrived. RFC 7230 4.1 makes + // them the framing, so a close before them is a truncated body, however many + // whole chunks came first - reporting what did arrive would be reporting a + // prefix as the whole. + self.http.phase = .failed; + self.http.err = error.ConnectionClosed; + } else if (self.http.content_length) |len| { + if (self.http.out_len >= len) { + self.http.phase = .complete; + } else { + // Fewer body bytes than Content-Length promised. + self.http.phase = .failed; + self.http.err = error.ConnectionClosed; + } + } else { + // No Content-Length: the FIN *is* the framing (RFC 7230 3.3.3 case 7). + self.http.phase = .complete; + } + }, + .head => { + self.http.phase = .failed; + self.http.err = error.ConnectionClosed; + }, + else => {}, + } + } + + // ================================================================================ tick + + /// Advance time. Drives DHCP retransmission and renewal, ARP resolution and TCP + /// retransmission. `now_ms` must be monotonic; it need not start at zero and it need not be + /// called at any particular rate, but nothing times out between calls, so a 47-second TCP + /// deadline needs ticks more often than every 47 seconds to be observed on time. + pub fn tick(self: *Stack, now_ms: u64) void { + self.now_ms = now_ms; + // Stir. The MAC alone would make every boot draw the same transaction ids, initial sequence + // numbers and ephemeral ports, which is how two runs of the same firmware end up accepting + // each other's stale DHCP replies. `now_ms` is the only outside input this file has, and a + // caller that ticks a real timer before starting DHCP therefore gets a different sequence + // every boot. Still not a source of security-relevant randomness - see `entropy`. + self.entropy ^= now_ms *% 0x9e37_79b9_7f4a_7c15; + self.dhcpTick(); + self.dnsTick(); + self.tcpTick(); + } +}; + +// The footprint claim, enforced at compile time, so a buffer that grows fails the build rather than +// the board. +// +// 4 KiB is the budget and it is measured, not guessed: the image has ~128 KB of L2MEM, nothing +// initialises the 32 MB of PSRAM, and ESP-Hosted's queues and its seven task stacks are competing +// for the same space. `Stack` is 3,576 bytes today. The failure this prevents is a stack overflow +// on a part with no debugger, which is indistinguishable from the SDIO bus not coming up. +comptime { + // 6 KiB, raised from 4 KiB when `http_head_max` went from 1024 to 2048 to fit a real CDN + // response head (1043 bytes measured). This is a regression alarm, not a budget: it exists so a + // buffer cannot grow unnoticed, and moving it is a decision to be justified at the buffer that + // caused it - which the comment on `http_head_max` does. The image's real constraint is the + // ~128 KB of L2MEM, and the heap in examples/http.zig was reduced by the same amount to pay for + // this. + if (Stack.footprint > 6 * 1024) @compileError(std.fmt.comptimePrint( + "ip.Stack is {d} bytes, over the 4 KiB budget", + .{Stack.footprint}, + )); +} + +// The host tests live in `ip_test.zig` - 117 cases, and they are the correctness argument for this +// slice, since it is the one part of the P4 bring-up that can be proven without the board. They are +// in their own file because they are longer than the stack, and because the tests deliberately +// re-derive every header offset from the RFCs rather than importing the tables above: a test that +// shares the constant it is checking passes on a consistent misreading. +// +// This reference is what makes `zig build test` find them: build.zig runs `src/net/ip.zig` as a +// test root, and Zig only collects tests from files the root actually references. +test { + _ = @import("ip_test.zig"); +} diff --git a/src/net/ip_test.zig b/src/net/ip_test.zig new file mode 100644 index 0000000..3e4c1f7 --- /dev/null +++ b/src/net/ip_test.zig @@ -0,0 +1,3029 @@ +//! Host tests for the IPv4 stack. +//! +//! This is the one slice of the P4 bring-up that can be *proven* without the board, and this file is +//! the proof. The stack takes frames through `onFrame` and time through `tick`, so a network here is +//! a function that writes bytes by hand and reads back whatever the stack handed to its `send` +//! callback. Nothing is mocked, nothing is stubbed: the code under test is the code that will run on +//! the die, byte for byte. +//! +//! Two rules keep this honest: +//! +//! * **The headers are re-derived here.** These tests do not import `ip.zig`'s offset tables; they +//! write literal offsets taken from the RFCs and from lwIP's packed structs. A test that shared +//! the constant it was checking would pass on a consistent misreading of the RFC, which is +//! exactly the failure mode this stack has to avoid. Where the two transcriptions disagree, one +//! of them is wrong and the test says so. +//! * **Every checksum is verified, never merely computed.** A checksum built by the same helper +//! the stack uses would prove nothing. `verify` below sums the received bytes independently and +//! asserts the fold is zero, which is the property a peer's kernel will check. +//! +//! Run with: zig build test + +const std = @import("std"); +const testing = std.testing; +const ip = @import("ip.zig"); + +// ============================================================================ capture rig +// +// `Stack.init` takes `*const fn ([]const u8) void` - no context pointer - so the captured frames +// have to live somewhere a plain function can reach. That is a wart in the interface, not in the +// stack, and the cost is this file-scope buffer. + +const cap_max = 32; +var cap_bytes: [cap_max][ip.frame_max]u8 = undefined; +var cap_lens: [cap_max]usize = undefined; +var cap_n: usize = 0; +var cap_over: usize = 0; + +fn capture(frame: []const u8) void { + if (cap_n == cap_max) { + cap_over += 1; + return; + } + @memcpy(cap_bytes[cap_n][0..frame.len], frame); + cap_lens[cap_n] = frame.len; + cap_n += 1; +} + +fn clearCapture() void { + cap_n = 0; + cap_over = 0; +} + +fn sent(i: usize) []const u8 { + return cap_bytes[i][0..cap_lens[i]]; +} + +fn lastSent() []const u8 { + return sent(cap_n - 1); +} + +/// A stack with a MAC and an empty capture log. Every test starts here. +fn newStack() ip.Stack { + clearCapture(); + return .init(our_mac, capture); +} + +const our_mac: ip.Mac = .{ 0x40, 0x4c, 0xca, 0xfe, 0x00, 0x01 }; +const gw_mac: ip.Mac = .{ 0x02, 0x00, 0x00, 0x11, 0x22, 0x33 }; +const peer_mac: ip.Mac = .{ 0x02, 0x00, 0x00, 0xaa, 0xbb, 0xcc }; +const our_ip: ip.Ip4 = .{ 192, 168, 1, 42 }; +const gw_ip: ip.Ip4 = .{ 192, 168, 1, 1 }; +const mask24: ip.Ip4 = .{ 255, 255, 255, 0 }; +const peer_ip: ip.Ip4 = .{ 192, 168, 1, 90 }; +const off_net_ip: ip.Ip4 = .{ 93, 184, 216, 34 }; +const bcast_mac: ip.Mac = .{ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff }; +/// RFC 826: the target hardware address of a request is "don't care". +const zero_mac: ip.Mac = .{ 0, 0, 0, 0, 0, 0 }; + +// ================================================================== independent primitives +// +// Header offsets written out again, from the RFCs. See the note at the top of the file. + +/// RFC 1071. Written differently from `ip.Checksum` on purpose: a `u32` accumulator over +/// `readInt`-free manual pairing, so a mistake in one is not a mistake in both. +fn sum16(bytes: []const u8) u32 { + var s: u32 = 0; + var i: usize = 0; + while (i + 1 < bytes.len) : (i += 2) { + s += (@as(u32, bytes[i]) << 8) | bytes[i + 1]; + } + if (i < bytes.len) s += @as(u32, bytes[i]) << 8; + while (s >> 16 != 0) s = (s & 0xffff) + (s >> 16); + return s; +} + +/// The property every receiver relies on: a buffer that already contains its own checksum sums to +/// 0xffff, so the complement is zero. +fn verify(bytes: []const u8) !void { + try testing.expectEqual(@as(u32, 0xffff), sum16(bytes)); +} + +fn verifyTransport(src: ip.Ip4, dst: ip.Ip4, proto: u8, seg: []const u8) !void { + var ph: [12]u8 = undefined; + @memcpy(ph[0..4], &src); + @memcpy(ph[4..8], &dst); + ph[8] = 0; + ph[9] = proto; + std.mem.writeInt(u16, ph[10..12], @intCast(seg.len), .big); + var s = sum16(&ph) + sum16(seg); + while (s >> 16 != 0) s = (s & 0xffff) + (s >> 16); + try testing.expectEqual(@as(u32, 0xffff), s); +} + +fn be16(b: []const u8, off: usize) u16 { + return std.mem.readInt(u16, b[off..][0..2], .big); +} +fn be32(b: []const u8, off: usize) u32 { + return std.mem.readInt(u32, b[off..][0..4], .big); +} +fn put16(b: []u8, off: usize, v: u16) void { + std.mem.writeInt(u16, b[off..][0..2], v, .big); +} +fn put32(b: []u8, off: usize, v: u32) void { + std.mem.writeInt(u32, b[off..][0..4], v, .big); +} + +/// A scratch frame under construction. `len` is the total frame length. +const Frame = struct { + buf: [ip.frame_max]u8 = undefined, + len: usize = 0, + + /// Ethernet II: 6 destination, 6 source, 2 ethertype. RFC 894 / lwIP `prot/ethernet.h:76-83`. + fn eth(self: *Frame, dst: ip.Mac, src: ip.Mac, ethertype: u16) void { + @memcpy(self.buf[0..6], &dst); + @memcpy(self.buf[6..12], &src); + put16(&self.buf, 12, ethertype); + self.len = 14; + } + + /// RFC 791 3.1. Fills the header and returns the payload slice to be written; the caller then + /// calls `sealIp`. + fn ip4(self: *Frame, src: ip.Ip4, dst: ip.Ip4, proto: u8, payload_len: usize) []u8 { + const h = self.buf[14..][0..20]; + h[0] = 0x45; + h[1] = 0; + put16(h, 2, @intCast(20 + payload_len)); + put16(h, 4, 0x1234); + put16(h, 6, 0); + h[8] = 64; + h[9] = proto; + put16(h, 10, 0); + @memcpy(h[12..16], &src); + @memcpy(h[16..20], &dst); + self.len = 14 + 20 + payload_len; + return self.buf[34 .. 34 + payload_len]; + } + + fn sealIp(self: *Frame) void { + const h = self.buf[14..][0..20]; + put16(h, 10, 0); + put16(h, 10, ~@as(u16, @truncate(sum16(h)))); + } + + /// Fill in a UDP or TCP checksum over the pseudo-header plus the segment. + fn sealTransport(self: *Frame, chksum_off: usize) void { + const h = self.buf[14..][0..20]; + const proto = h[9]; + const seg = self.buf[34..self.len]; + var ph: [12]u8 = undefined; + @memcpy(ph[0..4], h[12..16]); + @memcpy(ph[4..8], h[16..20]); + ph[8] = 0; + ph[9] = proto; + std.mem.writeInt(u16, ph[10..12], @intCast(seg.len), .big); + put16(seg, chksum_off, 0); + var s = sum16(&ph) + sum16(seg); + while (s >> 16 != 0) s = (s & 0xffff) + (s >> 16); + put16(seg, chksum_off, ~@as(u16, @truncate(s))); + self.sealIp(); + } + + fn bytes(self: *const Frame) []const u8 { + return self.buf[0..self.len]; + } +}; + +/// RFC 826 packet format, 28 bytes. lwIP `prot/etharp.h:86-96`. +fn arpFrame(opcode: u16, sha: ip.Mac, spa: ip.Ip4, tha: ip.Mac, tpa: ip.Ip4, eth_dst: ip.Mac) Frame { + var f: Frame = .{}; + f.eth(eth_dst, sha, 0x0806); + const a = f.buf[14..][0..28]; + put16(a, 0, 1); // hwtype: Ethernet + put16(a, 2, 0x0800); // proto: IPv4 + a[4] = 6; + a[5] = 4; + put16(a, 6, opcode); + @memcpy(a[8..14], &sha); + @memcpy(a[14..18], &spa); + @memcpy(a[18..24], &tha); + @memcpy(a[24..28], &tpa); + f.len = 14 + 28; + return f; +} + +/// RFC 792 echo. `payload` is the data after the 8-byte header. +fn icmpEchoFrame(src: ip.Ip4, dst: ip.Ip4, id: u16, seq: u16, payload: []const u8) Frame { + var f: Frame = .{}; + f.eth(our_mac, peer_mac, 0x0800); + const p = f.ip4(src, dst, 1, 8 + payload.len); + p[0] = 8; // echo request + p[1] = 0; + put16(p, 2, 0); + put16(p, 4, id); + put16(p, 6, seq); + @memcpy(p[8..], payload); + // ICMP has no pseudo-header (RFC 792): the checksum covers the message alone. + put16(p, 2, ~@as(u16, @truncate(sum16(p)))); + f.sealIp(); + return f; +} + +// ================================================================================= checksum + +test "RFC 1071 worked example" { + // RFC 1071 section 3, the byte sequence spelled out in the document's own figure: + // 00 01 f2 03 f4 f5 f6 f7 -> sum ddf2, checksum 220d + const data = [_]u8{ 0x00, 0x01, 0xf2, 0x03, 0xf4, 0xf5, 0xf6, 0xf7 }; + try testing.expectEqual(@as(u32, 0xddf2), sum16(&data)); + try testing.expectEqual(@as(u16, 0x220d), ip.checksum(&data)); +} + +test "checksum: incremental feeding matches contiguous, including at odd boundaries" { + // The bug this catches is a chunk of odd length leaving the high byte of a word unaccounted + // for. Splitting at every possible offset is cheap and total. + const data = [_]u8{ 0x45, 0x00, 0x00, 0x54, 0xab, 0xcd, 0x40, 0x00, 0x40, 0x01, 0x00, 0x00, 0xc0, 0xa8, 0x01, 0x2a, 0xc0, 0xa8, 0x01, 0x01, 0x7f }; + const want = ip.checksum(&data); + var split: usize = 0; + while (split <= data.len) : (split += 1) { + var c: ip.Checksum = .{}; + c.update(data[0..split]); + c.update(data[split..]); + try testing.expectEqual(want, c.final()); + } + // Three-way split too, so two consecutive odd chunks are exercised. + var i: usize = 0; + while (i < data.len) : (i += 1) { + var j: usize = i; + while (j < data.len) : (j += 1) { + var c: ip.Checksum = .{}; + c.update(data[0..i]); + c.update(data[i..j]); + c.update(data[j..]); + try testing.expectEqual(want, c.final()); + } + } +} + +test "checksum: an odd-length buffer is padded with a zero byte, not with the previous byte" { + // RFC 1071 section 1. A three-byte buffer must checksum as if it were four with a trailing 0. + const odd = [_]u8{ 0xde, 0xad, 0xbe }; + const padded = [_]u8{ 0xde, 0xad, 0xbe, 0x00 }; + try testing.expectEqual(ip.checksum(&padded), ip.checksum(&odd)); +} + +test "checksum: an all-zero buffer checksums to 0xffff, never to 0x0000" { + // A transmitted zero means "no checksum" in UDP, so the distinction is load-bearing. + const zeros: [20]u8 = @splat(0); + try testing.expectEqual(@as(u16, 0xffff), ip.checksum(&zeros)); +} + +test "checksum: RFC 768's transmitted zero is sent as 0xffff" { + // A UDP checksum field of zero means "not computed", so a datagram whose checksum genuinely + // works out to zero must transmit the arithmetically equivalent 0xffff instead. Tested on the + // helper because the case cannot be provoked by choosing DHCP option bytes: it depends on the + // whole datagram, headers included, summing to exactly 0xffff. + try testing.expectEqual(@as(u16, 0xffff), ip.udpChecksumOnWire(0)); + try testing.expectEqual(@as(u16, 0xffff), ip.udpChecksumOnWire(0xffff)); + try testing.expectEqual(@as(u16, 0x1234), ip.udpChecksumOnWire(0x1234)); +} + +test "checksum: a real IPv4 header verifies to zero once its own checksum is in place" { + var h = [_]u8{ 0x45, 0x00, 0x00, 0x3c, 0x1c, 0x46, 0x40, 0x00, 0x40, 0x06, 0x00, 0x00, 0xac, 0x10, 0x0a, 0x63, 0xac, 0x10, 0x0a, 0x0c }; + const c = ip.checksum(&h); + put16(&h, 10, c); + try verify(&h); + // And the classic published value for this header, from the Wikipedia/Comer worked example. + try testing.expectEqual(@as(u16, 0xb1e6), c); +} + +// ====================================================================================== ARP + +test "ARP: a request for our address is answered, and the reply is well formed" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + // setStatic announces; drop that so the reply is the only frame under test. + clearCapture(); + + var req = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(req.bytes()); + + try testing.expectEqual(@as(usize, 1), cap_n); + const r = lastSent(); + try testing.expectEqual(@as(usize, 42), r.len); + // Unicast back to the requester, not broadcast: a broadcast reply is legal but wasteful, and + // every stack on the segment would have to parse it. + try testing.expectEqualSlices(u8, &peer_mac, r[0..6]); + try testing.expectEqualSlices(u8, &our_mac, r[6..12]); + try testing.expectEqual(@as(u16, 0x0806), be16(r, 12)); + + const a = r[14..42]; + try testing.expectEqual(@as(u16, 1), be16(a, 0)); // hwtype Ethernet + try testing.expectEqual(@as(u16, 0x0800), be16(a, 2)); // proto IPv4 + try testing.expectEqual(@as(u8, 6), a[4]); + try testing.expectEqual(@as(u8, 4), a[5]); + try testing.expectEqual(@as(u16, 2), be16(a, 6)); // reply + try testing.expectEqualSlices(u8, &our_mac, a[8..14]); // sender hw = us + try testing.expectEqualSlices(u8, &our_ip, a[14..18]); // sender proto = us + try testing.expectEqualSlices(u8, &peer_mac, a[18..24]); // target hw = requester + try testing.expectEqualSlices(u8, &peer_ip, a[24..28]); +} + +test "ARP: a request for somebody else's address is ignored" { + var s = newStack(); + s.setStatic(our_ip, mask24, gw_ip); + clearCapture(); + var req = arpFrame(1, peer_mac, peer_ip, zero_mac, .{ 192, 168, 1, 77 }, bcast_mac); + s.onFrame(req.bytes()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "ARP: a malformed header is rejected on all four RFC 826 reception checks" { + const bad_fields = [_]struct { off: usize, val: u8 }{ + .{ .off = 1, .val = 2 }, // hwtype 2, not Ethernet + .{ .off = 3, .val = 0x06 }, // proto 0x0806, not IPv4 + .{ .off = 4, .val = 8 }, // hwlen 8 + .{ .off = 5, .val = 16 }, // protolen 16 + }; + for (bad_fields) |bad| { + var s = newStack(); + s.setStatic(our_ip, mask24, gw_ip); + clearCapture(); + var req = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + req.buf[14 + bad.off] = bad.val; + s.onFrame(req.bytes()); + try testing.expectEqual(@as(usize, 0), cap_n); + } +} + +test "ARP: setStatic announces the address gratuitously" { + var s = newStack(); + s.tick(500); + s.setStatic(our_ip, mask24, gw_ip); + try testing.expectEqual(@as(usize, 1), cap_n); + const g = lastSent(); + try testing.expectEqualSlices(u8, &bcast_mac, g[0..6]); + try testing.expectEqual(@as(u16, 0x0806), be16(g, 12)); + const a = g[14..42]; + try testing.expectEqual(@as(u16, 1), be16(a, 6)); // a request... + try testing.expectEqualSlices(u8, &our_ip, a[14..18]); // ...whose sender... + try testing.expectEqualSlices(u8, &our_ip, a[24..28]); // ...and target are both us +} + +test "ARP: a four-entry cache is not thrashed by unrelated broadcast traffic" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + + // Learn the gateway the legitimate way: it ARPs for us, we reply, and it goes in the cache. + var probe = arpFrame(1, gw_mac, gw_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + + // Now flood the segment with ARP between other hosts. None of it is addressed to us, so none + // of it may evict the gateway. + var k: u8 = 0; + while (k < 20) : (k += 1) { + var noise = arpFrame( + 1, + .{ 0x02, 0, 0, 0, 0, k }, + .{ 192, 168, 1, 100 + k }, + zero_mac, + .{ 192, 168, 1, 200 }, + bcast_mac, + ); + s.onFrame(noise.bytes()); + } + clearCapture(); + + // If the gateway survived, a datagram to an off-net address goes straight out to `gw_mac` + // instead of provoking an ARP request. + var echo = icmpEchoFrame(gw_ip, our_ip, 1, 1, "x"); + s.onFrame(echo.bytes()); + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqual(@as(u16, 0x0800), be16(lastSent(), 12)); // IPv4, not an ARP request + try testing.expectEqualSlices(u8, &gw_mac, lastSent()[0..6]); +} + +test "ARP: a cache entry ages out even while it is being used" { + // The bug this pins: refreshing an entry's timestamp on every lookup. It looks harmless and it + // means an entry kept alive by our own traffic is never re-resolved, so a gateway whose MAC + // changes is never noticed. + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + + // Keep using the entry, all the way past the 300 s age limit. + var now: u64 = 1000; + while (now < 400_000) : (now += 10_000) { + s.tick(now); + clearCapture(); + var echo = icmpEchoFrame(peer_ip, our_ip, 1, 1, "x"); + s.onFrame(echo.bytes()); + try testing.expectEqual(@as(usize, 1), cap_n); + } + // Past the limit the entry is gone: the reply is dropped and an ARP request goes in its place. + try testing.expectEqual(@as(u16, 0x0806), be16(lastSent(), 12)); + try testing.expectEqualSlices(u8, &peer_ip, lastSent()[14 + 24 ..][0..4]); +} + +test "ARP: a host that changes its MAC is followed" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + + // Same address, new hardware: a replaced router, or a VRRP failover. + const new_mac: ip.Mac = .{ 0x02, 0x00, 0x00, 0xde, 0xad, 0x01 }; + var again = arpFrame(1, new_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(again.bytes()); + clearCapture(); + + var echo = icmpEchoFrame(peer_ip, our_ip, 1, 1, "x"); + s.onFrame(echo.bytes()); + try testing.expectEqualSlices(u8, &new_mac, lastSent()[0..6]); +} + +test "IPv4: a received header carrying options is parsed by its own length field" { + // `ping -R` and any router-alert path produce these. A parser that assumes 20 bytes reads the + // options as the ICMP header and answers nonsense - or, worse, answers with the checksum + // covering the wrong bytes. + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + // 24-byte header: 20 plus a 4-byte NOP,NOP,NOP,END option block. + var f: Frame = .{}; + f.eth(our_mac, peer_mac, 0x0800); + const total = 24 + 8 + 4; + const h = f.buf[14..][0..24]; + h[0] = 0x46; // IPv4, 6 words of header + h[1] = 0; + put16(h, 2, total); + put16(h, 4, 0x1234); + put16(h, 6, 0); + h[8] = 64; + h[9] = 1; // ICMP + put16(h, 10, 0); + @memcpy(h[12..16], &peer_ip); + @memcpy(h[16..20], &our_ip); + h[20] = 1; // NOP + h[21] = 1; + h[22] = 1; + h[23] = 0; // END + put16(h, 10, ~@as(u16, @truncate(sum16(h)))); + const m = f.buf[14 + 24 ..][0 .. 8 + 4]; + m[0] = 8; + m[1] = 0; + put16(m, 2, 0); + put16(m, 4, 0x0102); + put16(m, 6, 0x0304); + @memcpy(m[8..], "wxyz"); + put16(m, 2, ~@as(u16, @truncate(sum16(m)))); + f.len = 14 + total; + s.onFrame(f.bytes()); + + try testing.expectEqual(@as(usize, 1), cap_n); + const r = lastSent(); + // The reply is emitted with a plain 20-byte header - nothing here generates options - and the + // echoed id, sequence and data prove the request's payload was found at the right offset. + try testing.expectEqual(@as(u8, 0x45), r[14]); + try verify(r[14..34]); + const e = r[34..]; + try testing.expectEqual(@as(u8, 0), e[0]); + try testing.expectEqual(@as(u16, 0x0102), be16(e, 4)); + try testing.expectEqual(@as(u16, 0x0304), be16(e, 6)); + try testing.expectEqualStrings("wxyz", e[8..12]); + try verify(e); +} + +// ===================================================================================== ICMP + +test "ICMP: an echo request is answered with a correct echo reply" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + clearCapture(); + // Teach the stack the peer's MAC by having it ARP for us first. + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + // The payload `ping` sends: 56 bytes, a timestamp then a counting pattern. + var payload: [56]u8 = undefined; + for (&payload, 0..) |*b, i| b.* = @intCast(i); + var req = icmpEchoFrame(peer_ip, our_ip, 0xbeef, 7, &payload); + s.onFrame(req.bytes()); + + try testing.expectEqual(@as(usize, 1), cap_n); + const r = lastSent(); + try testing.expectEqual(@as(usize, 14 + 20 + 8 + 56), r.len); + try testing.expectEqualSlices(u8, &peer_mac, r[0..6]); + try testing.expectEqual(@as(u16, 0x0800), be16(r, 12)); + + const h = r[14..34]; + try testing.expectEqual(@as(u8, 0x45), h[0]); + try testing.expectEqual(@as(u16, 20 + 8 + 56), be16(h, 2)); + try testing.expectEqual(@as(u8, 1), h[9]); // ICMP + // RFC 1122 3.2.1.7 recommends 64. A TTL of 1 is the failure that works on the bench and dies + // at the first router, which is the worst possible time to find out. + try testing.expectEqual(@as(u8, 64), h[8]); + // Don't Fragment: this stack neither fragments nor reassembles, so a router must not fragment + // what it cannot rebuild. + try testing.expectEqual(@as(u16, 0x4000), be16(h, 6)); + try testing.expectEqualSlices(u8, &our_ip, h[12..16]); // src and dst swapped + try testing.expectEqualSlices(u8, &peer_ip, h[16..20]); + try verify(h); // the IP header checksum, checked independently + + const m = r[34..]; + try testing.expectEqual(@as(u8, 0), m[0]); // echo reply + try testing.expectEqual(@as(u8, 0), m[1]); + try testing.expectEqual(@as(u16, 0xbeef), be16(m, 4)); // id echoed + try testing.expectEqual(@as(u16, 7), be16(m, 6)); // sequence echoed + try testing.expectEqualSlices(u8, &payload, m[8..]); + try verify(m); // and the ICMP checksum +} + +test "ICMP: a request with a bad IP header checksum is dropped and counted" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + var req = icmpEchoFrame(peer_ip, our_ip, 1, 1, "abcd"); + req.buf[14 + 10] ^= 0xff; // corrupt the IP header checksum + s.onFrame(req.bytes()); + try testing.expectEqual(@as(usize, 0), cap_n); + try testing.expectEqual(@as(u32, 1), s.counters.checksum_bad); +} + +test "ICMP: a request with a bad ICMP checksum is dropped and counted" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + var req = icmpEchoFrame(peer_ip, our_ip, 1, 1, "abcd"); + req.buf[34 + 2] ^= 0xff; // corrupt the ICMP checksum + s.onFrame(req.bytes()); + try testing.expectEqual(@as(usize, 0), cap_n); + try testing.expectEqual(@as(u32, 1), s.counters.checksum_bad); +} + +test "ICMP: a fragment is dropped rather than answered as a whole datagram" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_mac, peer_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + var req = icmpEchoFrame(peer_ip, our_ip, 1, 1, "abcd"); + put16(&req.buf, 14 + 6, 0x2000); // MF set + req.sealIp(); + s.onFrame(req.bytes()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "a frame addressed to another station is dropped" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + clearCapture(); + var req = icmpEchoFrame(peer_ip, our_ip, 1, 1, "abcd"); + req.buf[0] = 0x02; // not our MAC, not broadcast + s.onFrame(req.bytes()); + try testing.expectEqual(@as(usize, 0), cap_n); + try testing.expect(s.counters.rx_dropped >= 1); +} + +// ===================================================================================== DHCP +// +// RFC 2131. The synthetic server below is what a real one does with the fields that matter, and +// nothing else: no relay agent, no overload, no vendor options. + +/// Offsets into the BOOTP message, from RFC 2131 figure 1 / lwIP `prot/dhcp.h:50-91`. +const d = struct { + const op = 0; + const htype = 1; + const hlen = 2; + const xid = 4; + const secs = 8; + const flags = 10; + const ciaddr = 12; + const yiaddr = 16; + const siaddr = 20; + const chaddr = 28; + const cookie = 236; + const options = 240; +}; + +fn dhcpReply(kind: u8, xid: u32, yiaddr: ip.Ip4, server: ip.Ip4, opts: []const u8, dst_ip: ip.Ip4, dst_mac: ip.Mac) Frame { + var f: Frame = .{}; + f.eth(dst_mac, gw_mac, 0x0800); + const payload_len = 8 + d.options + 3 + opts.len + 1; + const p = f.ip4(server, dst_ip, 17, payload_len); + put16(p, 0, 67); // source port: DHCP server + put16(p, 2, 68); // destination port: DHCP client + put16(p, 4, @intCast(payload_len)); + put16(p, 6, 0); + const m = p[8..]; + @memset(m, 0); + m[d.op] = 2; // BOOTREPLY + m[d.htype] = 1; + m[d.hlen] = 6; + put32(m, d.xid, xid); + @memcpy(m[d.yiaddr..][0..4], &yiaddr); + @memcpy(m[d.siaddr..][0..4], &server); + @memcpy(m[d.chaddr..][0..6], &our_mac); + put32(m, d.cookie, 0x63825363); + m[d.options] = 53; // message type + m[d.options + 1] = 1; + m[d.options + 2] = kind; + @memcpy(m[d.options + 3 ..][0..opts.len], opts); + m[d.options + 3 + opts.len] = 255; // END + f.sealTransport(6); + return f; +} + +/// Option 1 (mask), 3 (router), 6 (DNS), 51 (lease), 54 (server id) for the network in the brief. +const standard_opts = [_]u8{ + 1, 4, 255, 255, 255, 0, // subnet mask /24 + 3, 4, 192, 168, 1, 1, // router + 6, 4, 192, 168, 1, 1, // DNS + 51, 4, 0, 0, 0x1c, 0x20, // lease 7200 s + 54, 4, 192, 168, 1, 1, // server identifier +}; + +fn findOption(msg: []const u8, want: u8) ?[]const u8 { + var i: usize = d.options; + while (i < msg.len) { + if (msg[i] == 255) return null; + if (msg[i] == 0) { + i += 1; + continue; + } + if (i + 2 > msg.len) return null; + const len = msg[i + 1]; + if (i + 2 + len > msg.len) return null; + if (msg[i] == want) return msg[i + 2 ..][0..len]; + i += 2 + len; + } + return null; +} + +/// The DHCP message inside a captured frame, and a few sanity checks that apply to all of them. +fn dhcpOut(frame: []const u8) ![]const u8 { + try testing.expectEqual(@as(u16, 0x0800), be16(frame, 12)); + const h = frame[14..34]; + try testing.expectEqual(@as(u8, 17), h[9]); // UDP + try verify(h); + const seg = frame[34..]; + try testing.expectEqual(@as(u16, 68), be16(seg, 0)); // from the client port + try testing.expectEqual(@as(u16, 67), be16(seg, 2)); // to the server port + try testing.expectEqual(@as(u16, @intCast(seg.len)), be16(seg, 4)); + try verifyTransport(h[12..16].*, h[16..20].*, 17, seg); + const msg = seg[8..]; + try testing.expectEqual(@as(u8, 1), msg[d.op]); // BOOTREQUEST + try testing.expectEqual(@as(u8, 1), msg[d.htype]); // Ethernet + try testing.expectEqual(@as(u8, 6), msg[d.hlen]); + try testing.expectEqual(@as(u32, 0x63825363), be32(msg, d.cookie)); + try testing.expectEqualSlices(u8, &our_mac, msg[d.chaddr..][0..6]); + // RFC 951: a BOOTP message is at least 300 bytes. + try testing.expect(msg.len >= 300); + return msg; +} + +test "DHCP: a full DISCOVER / OFFER / REQUEST / ACK exchange binds the address" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); + + // ---- DISCOVER + try testing.expectEqual(@as(usize, 1), cap_n); + const disc_frame = sent(0); + // Broadcast at both layers: no address yet, so nothing else could work. + try testing.expectEqualSlices(u8, &bcast_mac, disc_frame[0..6]); + try testing.expectEqualSlices(u8, &.{ 0, 0, 0, 0 }, disc_frame[14 + 12 ..][0..4]); + try testing.expectEqualSlices(u8, &.{ 255, 255, 255, 255 }, disc_frame[14 + 16 ..][0..4]); + const disc = try dhcpOut(disc_frame); + try testing.expectEqual(@as(u16, 0x8000), be16(disc, d.flags)); // ask for a broadcast reply + try testing.expectEqualSlices(u8, &.{ 0, 0, 0, 0 }, disc[d.ciaddr..][0..4]); + try testing.expectEqualSlices(u8, &.{1}, findOption(disc, 53).?); // DHCPDISCOVER + try testing.expect(findOption(disc, 55) != null); // parameter request list + try testing.expect(findOption(disc, 57) != null); // maximum message size + // A DISCOVER must not claim an address or name a server. + try testing.expect(findOption(disc, 50) == null); + try testing.expect(findOption(disc, 54) == null); + const xid = be32(disc, d.xid); + + // ---- OFFER, unicast to the address about to be granted (RFC 2131 4.1 permits this, and it is + // the case that only works because `ip4Input` lets UDP through while unbound). + clearCapture(); + var offer = dhcpReply(2, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(offer.bytes()); + try testing.expectEqual(ip.DhcpState.requesting, s.dhcpState()); + + // ---- REQUEST + try testing.expectEqual(@as(usize, 1), cap_n); + const req = try dhcpOut(sent(0)); + try testing.expectEqual(xid, be32(req, d.xid)); // same transaction + try testing.expectEqualSlices(u8, &.{3}, findOption(req, 53).?); // DHCPREQUEST + // RFC 2131 4.3.2: SELECTING carries the offered address in option 50 and the server it is + // accepting in option 54, and `ciaddr` stays zero. + try testing.expectEqualSlices(u8, &our_ip, findOption(req, 50).?); + try testing.expectEqualSlices(u8, &gw_ip, findOption(req, 54).?); + try testing.expectEqualSlices(u8, &.{ 0, 0, 0, 0 }, req[d.ciaddr..][0..4]); + + // ---- ACK + clearCapture(); + var ack = dhcpReply(5, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(ack.bytes()); + + try testing.expectEqual(ip.DhcpState.bound, s.dhcpState()); + try testing.expectEqual(our_ip, s.ip().?); + try testing.expectEqual(mask24, s.netmask()); + try testing.expectEqual(gw_ip, s.gateway()); + try testing.expectEqual(gw_ip, s.dnsServer().?); + // Binding announces the new address. + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqual(@as(u16, 0x0806), be16(lastSent(), 12)); + try testing.expectEqualSlices(u8, &our_ip, lastSent()[14 + 14 ..][0..4]); +} + +test "DHCP: a reply with the wrong transaction id is ignored" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + var offer = dhcpReply(2, xid ^ 0xffff_ffff, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(offer.bytes()); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "DHCP: a reply for another station's hardware address is ignored" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + var offer = dhcpReply(2, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + offer.buf[34 + 8 + d.chaddr + 5] ^= 0xff; // a different chaddr + offer.sealTransport(6); + s.onFrame(offer.bytes()); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "DHCP: DISCOVER is retransmitted with a growing backoff and the same transaction id" { + var s = newStack(); + s.tick(0); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + + // Nothing before the first backoff expires. + s.tick(1_999); + try testing.expectEqual(@as(usize, 0), cap_n); + s.tick(2_000); + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqual(xid, be32(sent(0)[42..], d.xid)); + + // The next interval is longer: nothing at +2 s, a frame at +4 s. + s.tick(5_999); + try testing.expectEqual(@as(usize, 1), cap_n); + s.tick(6_000); + try testing.expectEqual(@as(usize, 2), cap_n); + + // And the `secs` field tracks how long acquisition has been going. + try testing.expectEqual(@as(u16, 6), be16(sent(1)[42..], d.secs)); +} + +test "DHCP: a NAK surrenders the address and restarts from DISCOVER" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + var offer = dhcpReply(2, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(offer.bytes()); + clearCapture(); + + var nak = dhcpReply(6, xid, .{ 0, 0, 0, 0 }, gw_ip, &.{}, ip.ip_broadcast, bcast_mac); + s.onFrame(nak.bytes()); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); + try testing.expect(s.ip() == null); + // And a fresh DISCOVER went out immediately. + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqualSlices(u8, &.{1}, findOption(try dhcpOut(sent(0)), 53).?); +} + +test "DHCP: at T1 the lease is renewed by unicast REQUEST with ciaddr set" { + var s = newStack(); + s.tick(0); + s.dhcpStart(); + const xid0 = be32(sent(0)[42..], d.xid); + var offer = dhcpReply(2, xid0, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(offer.bytes()); + var ack = dhcpReply(5, xid0, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(ack.bytes()); + try testing.expectEqual(ip.DhcpState.bound, s.dhcpState()); + + // Lease 7200 s, so T1 = 3600 s (lwIP `core/ipv4/dhcp.c:757`: half the lease). + clearCapture(); + s.tick(3_599_000); + try testing.expectEqual(@as(usize, 0), cap_n); + try testing.expectEqual(ip.DhcpState.bound, s.dhcpState()); + + // T1. The REQUEST is unicast to the server, so it needs the server's MAC first: with the cache + // empty, the datagram is dropped and an ARP request goes out in its place. + s.tick(3_600_000); + try testing.expectEqual(ip.DhcpState.renewing, s.dhcpState()); + try testing.expectEqual(@as(u16, 0x0806), be16(sent(0), 12)); + try testing.expectEqualSlices(u8, &gw_ip, sent(0)[14 + 24 ..][0..4]); // ARP for the server + + // The server answers by ARPing for us, which is enough to populate the cache. + var probe = arpFrame(1, gw_mac, gw_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + // The next retransmission now has a route. + s.tick(3_602_000); + try testing.expectEqual(@as(usize, 1), cap_n); + const r = lastSent(); + try testing.expectEqualSlices(u8, &gw_mac, r[0..6]); // unicast to the server + try testing.expectEqualSlices(u8, &gw_ip, r[14 + 16 ..][0..4]); + const msg = try dhcpOut(r); + try testing.expectEqualSlices(u8, &.{3}, findOption(msg, 53).?); // DHCPREQUEST + // RFC 2131 4.3.6, the RENEWING column: ciaddr carries the bound address, and there is no + // requested-IP option and no server identifier. + try testing.expectEqualSlices(u8, &our_ip, msg[d.ciaddr..][0..4]); + try testing.expect(findOption(msg, 50) == null); + try testing.expect(findOption(msg, 54) == null); + // A fresh transaction id for the new exchange (RFC 2131 4.4.5). + try testing.expect(be32(msg, d.xid) != xid0); + + // The server ACKs and the lease is extended from now. + const xid1 = be32(msg, d.xid); + clearCapture(); + var ack2 = dhcpReply(5, xid1, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(ack2.bytes()); + try testing.expectEqual(ip.DhcpState.bound, s.dhcpState()); + try testing.expectEqual(our_ip, s.ip().?); +} + +test "DHCP: at T2 renewal becomes a broadcast rebind, and an expired lease is surrendered" { + var s = newStack(); + s.tick(0); + s.dhcpStart(); + const xid0 = be32(sent(0)[42..], d.xid); + var offer = dhcpReply(2, xid0, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(offer.bytes()); + var ack = dhcpReply(5, xid0, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(ack.bytes()); + + // Give the stack the server's MAC so the renewal is not blocked on ARP. + var probe = arpFrame(1, gw_mac, gw_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + + s.tick(3_600_000); // T1 + try testing.expectEqual(ip.DhcpState.renewing, s.dhcpState()); + + // T2 = 7/8 of 7200 s = 6300 s (lwIP `core/ipv4/dhcp.c:766`). + clearCapture(); + s.tick(6_300_000); + try testing.expectEqual(ip.DhcpState.rebinding, s.dhcpState()); + try testing.expectEqual(@as(usize, 1), cap_n); + // Rebinding is broadcast: the granting server is not answering, so ask anybody. + try testing.expectEqualSlices(u8, &bcast_mac, lastSent()[0..6]); + const msg = try dhcpOut(lastSent()); + try testing.expectEqualSlices(u8, &our_ip, msg[d.ciaddr..][0..4]); + try testing.expect(findOption(msg, 54) == null); + + // Lease expiry: the address must go, because the server may already have handed it out. + clearCapture(); + s.tick(7_200_000); + try testing.expect(s.ip() == null); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); +} + +test "DHCP: an option whose length runs past the datagram does not read off the end" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + // Option 54 - the server identifier, which the OFFER handler actually looks for - claiming 200 + // bytes of a message with three left. Unchecked, that is a 200-byte read past the end of the + // frame, which is the classic DHCP parser bug and is reachable by any host on the segment. + var offer = dhcpReply(2, xid, our_ip, gw_ip, &[_]u8{ 54, 200, 192, 168 }, our_ip, our_mac); + offer.sealTransport(6); + s.onFrame(offer.bytes()); + // The option did not resolve, so the handler fell back to `siaddr` - and the exchange carried + // on rather than crashing. + try testing.expectEqual(ip.DhcpState.requesting, s.dhcpState()); + try testing.expectEqual(@as(usize, 1), cap_n); + const req = try dhcpOut(sent(0)); + try testing.expectEqualSlices(u8, &gw_ip, findOption(req, 54).?); // from siaddr +} + +test "DHCP: an option truncated by one byte does not read off the end" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + // Length 4 with only three bytes of message left after it, counting the END marker. + var offer = dhcpReply(2, xid, our_ip, gw_ip, &[_]u8{ 54, 4, 192, 168 }, our_ip, our_mac); + offer.sealTransport(6); + s.onFrame(offer.bytes()); + try testing.expectEqual(ip.DhcpState.requesting, s.dhcpState()); +} + +test "DHCP: a bogus option before a good one does not hide it" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + // A zero-length option, then a pad, then the real server identifier. + var offer = dhcpReply(2, xid, our_ip, gw_ip, &[_]u8{ 12, 0, 0, 54, 4, 192, 168, 1, 1 }, our_ip, our_mac); + offer.sealTransport(6); + s.onFrame(offer.bytes()); + const req = try dhcpOut(sent(0)); + try testing.expectEqualSlices(u8, &gw_ip, findOption(req, 54).?); +} + +/// Cut `drop` bytes off the end of a UDP datagram and re-seal, so the last byte of the options is +/// wherever the caller wants it. `dhcpReply` always writes an END marker, and END is what stops a +/// well-behaved option walk - so the only way to test what happens when the walk reaches the end of +/// the buffer instead is to take the marker away. +fn truncateUdp(f: *Frame, drop: usize) void { + f.len -= drop; + const h = f.buf[14..][0..20]; + put16(h, 2, @intCast(f.len - 14)); + const seg = f.buf[34..f.len]; + put16(seg, 4, @intCast(seg.len)); + f.sealTransport(6); +} + +test "DHCP: an option code in the last byte, with no length byte after it, is not read past" { + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + // A hostname option, then a bare code 3 where a length byte should be. The END marker that + // `dhcpReply` appends is cut off, so the walk runs into the end of the datagram - and no + // option 54 is present, so the handler's search for the server identifier walks the whole + // list and reaches that last byte. Unchecked, reading its length byte is one past the frame. + var offer = dhcpReply(2, xid, our_ip, gw_ip, &[_]u8{ 12, 1, 'x', 3 }, our_ip, our_mac); + truncateUdp(&offer, 1); + s.onFrame(offer.bytes()); + // It read what it could and stopped, and fell back to `siaddr` for the server identifier. + try testing.expectEqual(ip.DhcpState.requesting, s.dhcpState()); + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqualSlices(u8, &gw_ip, findOption(try dhcpOut(sent(0)), 54).?); +} + +test "DHCP: a reply without the magic cookie is not a DHCP message" { + // RFC 2131 3: the four-byte cookie is what distinguishes a DHCP message from plain BOOTP. + // Without the check, any BOOTP reply - or any UDP datagram to port 68 that happens to have the + // right xid in the right place - is parsed as options. + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + var offer = dhcpReply(2, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + put32(&offer.buf, 34 + 8 + d.cookie, 0x63825364); // one off + offer.sealTransport(6); + s.onFrame(offer.bytes()); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "DHCP: a BOOTREQUEST is not mistaken for a reply" { + // Every DISCOVER on the segment is a broadcast, including our own. A client that does not check + // the `op` field parses its own request - or another client's - as an offer, and RFC 2131 gives + // it a `yiaddr` of zero to work with. + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const xid = be32(sent(0)[42..], d.xid); + clearCapture(); + var offer = dhcpReply(2, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + offer.buf[34 + 8 + d.op] = 1; // BOOTREQUEST + offer.sealTransport(6); + s.onFrame(offer.bytes()); + try testing.expectEqual(ip.DhcpState.selecting, s.dhcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +// ====================================================================================== TCP +// +// The synthetic peer. Sequence numbers here are the *peer's*; the stack's are read out of what it +// sends, because its ISN is not something a test may assume. + +/// Offsets into the TCP header, RFC 793 3.1 / lwIP `prot/tcp.h:56-65`. +const t = struct { + const src = 0; + const dst = 2; + const seq = 4; + const ack = 8; + const hdrlen_flags = 12; + const window = 14; + const chksum = 16; + + const fin: u8 = 0x01; + const syn: u8 = 0x02; + const rst: u8 = 0x04; + const psh: u8 = 0x08; + const ack_f: u8 = 0x10; +}; + +const Peer = struct { + ip: ip.Ip4, + port: u16, + mac: ip.Mac, + /// Our own sequence space, as the peer. + seq: u32 = 0x1000_0000, + /// The stack's ports and sequence numbers, learnt from its SYN. + stack_port: u16 = 0, + window: u16 = 8192, + /// With an MSS option in our SYN-ACK, or without. + mss: ?u16 = 1460, + + fn segment(self: *Peer, flags: u8, ackno: u32, data: []const u8, with_mss: bool) Frame { + var f: Frame = .{}; + f.eth(our_mac, self.mac, 0x0800); + const opt_len: usize = if (with_mss) 4 else 0; + const p = f.ip4(self.ip, our_ip, 6, 20 + opt_len + data.len); + put16(p, t.src, self.port); + put16(p, t.dst, self.stack_port); + put32(p, t.seq, self.seq); + put32(p, t.ack, ackno); + put16(p, t.hdrlen_flags, (@as(u16, @intCast((20 + opt_len) / 4)) << 12) | flags); + put16(p, t.window, self.window); + put16(p, t.chksum, 0); + put16(p, 18, 0); + if (with_mss) { + p[20] = 2; + p[21] = 4; + put16(p, 22, self.mss.?); + } + if (data.len != 0) @memcpy(p[20 + opt_len ..], data); + f.sealTransport(t.chksum); + return f; + } +}; + +/// A captured TCP segment, decoded, with its checksums verified independently. +const Seg = struct { + src_port: u16, + dst_port: u16, + seq: u32, + ack: u32, + flags: u8, + window: u16, + data: []const u8, + mss: ?u16, +}; + +fn decode(frame: []const u8) !Seg { + try testing.expectEqual(@as(u16, 0x0800), be16(frame, 12)); + const h = frame[14..34]; + try testing.expectEqual(@as(u8, 6), h[9]); + try verify(h); + const total = be16(h, 2); + const seg = frame[34 .. 14 + total]; + try verifyTransport(h[12..16].*, h[16..20].*, 6, seg); + const hf = be16(seg, t.hdrlen_flags); + const hlen = @as(usize, hf >> 12) * 4; + var mss: ?u16 = null; + var i: usize = 20; + while (i + 1 < hlen) { + if (seg[i] == 0) break; + if (seg[i] == 1) { + i += 1; + continue; + } + const olen = seg[i + 1]; + if (olen < 2 or i + olen > hlen) break; + if (seg[i] == 2 and olen == 4) mss = be16(seg, i + 2); + i += olen; + } + return .{ + .src_port = be16(seg, t.src), + .dst_port = be16(seg, t.dst), + .seq = be32(seg, t.seq), + .ack = be32(seg, t.ack), + .flags = @truncate(hf & 0x3f), + .window = be16(seg, t.window), + .data = seg[hlen..], + .mss = mss, + }; +} + +/// Bring a stack up statically with the peer's MAC already in the ARP cache, then start a GET. +/// Returns the peer and the SYN the stack sent. +fn startGet(s: *ip.Stack, peer: *Peer, path: []const u8, out: []u8) !Seg { + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer.mac, peer.ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, path, out)); + try testing.expectEqual(@as(usize, 1), cap_n); + const syn = try decode(sent(0)); + peer.stack_port = syn.src_port; + return syn; +} + +/// Complete the handshake: deliver the SYN-ACK and return the sequence number that acknowledges the +/// whole request. Afterwards `sent(0)` is the request segment - the capture log is cleared first, so +/// tests never have to remember whether the SYN is still in it. That off-by-one is exactly the kind +/// of thing a test helper exists to remove. +fn handshake(s: *ip.Stack, peer: *Peer, iss: u32) !u32 { + clearCapture(); + var synack = peer.segment(t.syn | t.ack_f, iss +% 1, &.{}, true); + s.onFrame(synack.bytes()); + peer.seq +%= 1; + const req = try decode(sent(0)); + try testing.expect(req.data.len > 0); + return iss +% 1 +% @as(u32, @intCast(req.data.len)); +} + +test "TCP: the SYN offers an MSS, uses an ephemeral port and advertises a window" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + + try testing.expectEqual(t.syn, syn.flags); + try testing.expectEqual(@as(u16, 80), syn.dst_port); + try testing.expect(syn.src_port >= 49152); // RFC 6335 dynamic range + try testing.expectEqual(@as(?u16, 1460), syn.mss); + try testing.expect(syn.window > 0); + try testing.expectEqual(@as(usize, 0), syn.data.len); + try testing.expectEqual(ip.TcpState.syn_sent, s.tcpState()); +} + +test "TCP: a handshake, the request, a response and a clean teardown" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/index.html", &out); + const iss = syn.seq; + + // ---- SYN-ACK + _ = try handshake(&s, &peer, iss); + try testing.expectEqual(ip.TcpState.established, s.tcpState()); + + // The handshake's ACK carries the request: one frame, not two. + try testing.expectEqual(@as(usize, 1), cap_n); + const req = try decode(sent(0)); + try testing.expectEqual(t.ack_f | t.psh, req.flags); + try testing.expectEqual(iss +% 1, req.seq); + try testing.expectEqual(peer.seq, req.ack); + try testing.expect(std.mem.startsWith(u8, req.data, "GET /index.html HTTP/1.1\r\n")); + // The Host header is the address literal - there is no DNS here - and port 80 is elided. + try testing.expect(std.mem.indexOf(u8, req.data, "\r\nHost: 192.168.1.90\r\n") != null); + // Connection: close is the framing for a body with no Content-Length. + try testing.expect(std.mem.indexOf(u8, req.data, "\r\nConnection: close\r\n") != null); + try testing.expect(std.mem.endsWith(u8, req.data, "\r\n\r\n")); + const req_len = req.data.len; + + // ---- the peer acknowledges the request and sends the whole response in one segment + clearCapture(); + const body = "hello, world"; + const response = "HTTP/1.1 200 OK\r\nServer: test\r\nContent-Length: 12\r\n\r\n" ++ body; + var resp = peer.segment(t.ack_f | t.psh, iss +% 1 +% @as(u32, @intCast(req_len)), response, false); + s.onFrame(resp.bytes()); + peer.seq +%= @intCast(response.len); + + // The body is complete, so the stack half-closes: the FIN is the acknowledgement too. + try testing.expectEqual(@as(usize, 1), cap_n); + const fin = try decode(sent(0)); + try testing.expectEqual(t.fin | t.ack_f, fin.flags); + try testing.expectEqual(peer.seq, fin.ack); + try testing.expectEqual(ip.TcpState.fin_wait_1, s.tcpState()); + + // ---- the peer acknowledges our FIN and sends its own + clearCapture(); + var peer_fin = peer.segment(t.fin | t.ack_f, fin.seq +% 1, &.{}, false); + s.onFrame(peer_fin.bytes()); + peer.seq +%= 1; + const last = try decode(lastSent()); + try testing.expectEqual(t.ack_f, last.flags); + try testing.expectEqual(peer.seq, last.ack); + try testing.expectEqual(ip.TcpState.time_wait, s.tcpState()); + + // ---- and the body comes out + const n = try s.httpGet(peer.ip, peer.port, "/index.html", &out); + try testing.expectEqual(@as(usize, 12), n); + try testing.expectEqualStrings(body, out[0..n]); + try testing.expectEqual(@as(u16, 200), s.httpStatus()); + + // TIME_WAIT is short by design; it ends on the clock, not on a frame. + s.tick(1_000_000); + try testing.expectEqual(ip.TcpState.closed, s.tcpState()); +} + +test "TCP: the SYN is retransmitted with its MSS option, on a doubling timer" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + clearCapture(); + + // Nothing before the RTO. + s.tick(1_999); + try testing.expectEqual(@as(usize, 0), cap_n); + s.tick(2_000); + try testing.expectEqual(@as(usize, 1), cap_n); + const again = try decode(sent(0)); + try testing.expectEqual(t.syn, again.flags); + try testing.expectEqual(syn.seq, again.seq); + // The MSS option must be repeated: a peer that only ever sees the retransmission would + // otherwise fall back to 536. + try testing.expectEqual(@as(?u16, 1460), again.mss); + + // The next timeout is twice as long: 2 s, not 1 s. + s.tick(3_999); + try testing.expectEqual(@as(usize, 1), cap_n); + s.tick(4_000); + try testing.expectEqual(@as(usize, 2), cap_n); + try testing.expectEqual(@as(u32, 2), s.counters.tcp_retx); +} + +test "TCP: retransmission after a dropped data segment resends the identical bytes" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/drop", &out); + const iss = syn.seq; + + _ = try handshake(&s, &peer, iss); + const first = try decode(sent(0)); + try testing.expect(first.data.len > 0); + + // Pretend the segment was lost: never acknowledge it, just let time pass. + clearCapture(); + s.tick(1_500); + try testing.expectEqual(@as(usize, 0), cap_n); // handshake completed at t=1000, RTO at t=2000 + s.tick(2_000); + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqual(@as(u32, 1), s.counters.tcp_retx); + + const again = try decode(sent(0)); + try testing.expectEqual(first.seq, again.seq); + try testing.expectEqualSlices(u8, first.data, again.data); + try testing.expectEqual(first.flags, again.flags); + + // Now it gets through, and the connection carries on from the same place. + clearCapture(); + const response = "HTTP/1.1 204 No Content\r\nContent-Length: 0\r\n\r\n"; + var resp = peer.segment(t.ack_f, iss +% 1 +% @as(u32, @intCast(first.data.len)), response, false); + s.onFrame(resp.bytes()); + try testing.expectEqual(@as(usize, 0), try s.httpGet(peer.ip, peer.port, "/drop", &out)); + try testing.expectEqual(@as(u16, 204), s.httpStatus()); +} + +test "TCP: retransmission eventually gives up with TimedOut" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + _ = try startGet(&s, &peer, "/", &out); + + // Six retransmissions with a doubling, capped backoff, then failure. Ticking well past every + // deadline in one step is enough: the deadline is absolute. + var now: u64 = 1000; + var k: usize = 0; + while (k < 8) : (k += 1) { + now += 60_000; + s.tick(now); + } + try testing.expectEqual(ip.TcpState.closed, s.tcpState()); + try testing.expectError(error.TimedOut, s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqual(@as(u32, 6), s.counters.tcp_retx); +} + +test "TCP: an out-of-order segment is not accepted, and provokes a duplicate ACK" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + const iss = syn.seq; + const our_next = try handshake(&s, &peer, iss); + const in_order_seq = peer.seq; + + // The second half of the response arrives first. + const head = "HTTP/1.1 200 OK\r\nContent-Length: 4\r\n\r\n"; + clearCapture(); + peer.seq = in_order_seq +% @as(u32, @intCast(head.len)); + var late = peer.segment(t.ack_f, our_next, "abcd", false); + s.onFrame(late.bytes()); + + // A duplicate ACK for what we are still waiting for, and nothing consumed. + try testing.expectEqual(@as(usize, 1), cap_n); + const dup = try decode(sent(0)); + try testing.expectEqual(t.ack_f, dup.flags); + try testing.expectEqual(in_order_seq, dup.ack); + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, "/", &out)); + + // The missing piece arrives. + clearCapture(); + peer.seq = in_order_seq; + var missing = peer.segment(t.ack_f, our_next, head, false); + s.onFrame(missing.bytes()); + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqual(@as(u16, 200), s.httpStatus()); + + // And the retransmission of the tail completes it. + peer.seq = in_order_seq +% @as(u32, @intCast(head.len)); + var tail = peer.segment(t.ack_f, our_next, "abcd", false); + s.onFrame(tail.bytes()); + try testing.expectEqual(@as(usize, 4), try s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqualStrings("abcd", out[0..4]); +} + +test "TCP: a retransmission overlapping data already received is trimmed, not rejected" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + const iss = syn.seq; + const our_next = try handshake(&s, &peer, iss); + + // Headers first, so the overlap lands squarely in the body where duplicated bytes cannot hide + // in a header line the parser would have skipped anyway. + const head = "HTTP/1.1 200 OK\r\nContent-Length: 16\r\n\r\n"; + var h = peer.segment(t.ack_f, our_next, head, false); + s.onFrame(h.bytes()); + peer.seq +%= @intCast(head.len); + const base = peer.seq; + + // Ten body bytes. + var a = peer.segment(t.ack_f, our_next, "0123456789", false); + s.onFrame(a.bytes()); + + // Then a retransmission that starts four bytes before what we now expect and carries six new + // bytes past it. Without trimming, `6789` is written twice, `rcv_nxt` runs four ahead of the + // truth, and the final six bytes are then rejected as old - so the request never completes. + peer.seq = base +% 6; + var b = peer.segment(t.ack_f, our_next, "6789abcdef", false); + s.onFrame(b.bytes()); + + try testing.expectEqual(@as(usize, 16), try s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqualStrings("0123456789abcdef", out[0..16]); +} + +test "TCP: a SYN-ACK that does not acknowledge our SYN is reset, not accepted" { + // RFC 793 3.4: an old duplicate SYN-ACK, or one aimed at a previous incarnation of this + // 4-tuple, is answered with a reset. Accepting it would establish a connection whose sequence + // space the peer does not agree with, and every subsequent segment would be discarded. + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + clearCapture(); + + var wrong = peer.segment(t.syn | t.ack_f, syn.seq +% 999, &.{}, true); + s.onFrame(wrong.bytes()); + try testing.expectEqual(ip.TcpState.syn_sent, s.tcpState()); + try testing.expectEqual(@as(usize, 1), cap_n); + const rst = try decode(sent(0)); + try testing.expectEqual(t.rst, rst.flags); + try testing.expectEqual(syn.seq +% 999, rst.seq); // RST carries the offending ACK number + + // The right one still works. + clearCapture(); + var right = peer.segment(t.syn | t.ack_f, syn.seq +% 1, &.{}, true); + s.onFrame(right.bytes()); + try testing.expectEqual(ip.TcpState.established, s.tcpState()); +} + +test "TCP: a FIN ahead of the data we have is not honoured" { + // A FIN whose sequence number is past `rcv_nxt` closes the connection over a hole. Honouring it + // would report a complete body that is missing its middle. + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + const iss = syn.seq; + const our_next = try handshake(&s, &peer, iss); + const base = peer.seq; + + const head = "HTTP/1.1 200 OK\r\nContent-Length: 4\r\n\r\n"; + var h = peer.segment(t.ack_f, our_next, head, false); + s.onFrame(h.bytes()); + peer.seq +%= @intCast(head.len); + + // A FIN 100 bytes into the future, as though a segment we never saw preceded it. + clearCapture(); + peer.seq = base +% @as(u32, @intCast(head.len)) +% 100; + var early = peer.segment(t.fin | t.ack_f, our_next, &.{}, false); + s.onFrame(early.bytes()); + // Not closed, not completed: the body is still outstanding. + try testing.expectEqual(ip.TcpState.established, s.tcpState()); + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, "/", &out)); + + // The real body arrives and completes it. + peer.seq = base +% @as(u32, @intCast(head.len)); + var body = peer.segment(t.ack_f, our_next, "wxyz", false); + s.onFrame(body.bytes()); + try testing.expectEqual(@as(usize, 4), try s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqualStrings("wxyz", out[0..4]); +} + +test "TCP: an in-window RST tears the connection down; an out-of-window one does not" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + const iss = syn.seq; + _ = try handshake(&s, &peer, iss); + + // RFC 5961 3: a RST whose sequence number is not the next one expected gets a challenge ACK + // and is otherwise ignored. This is what stops a blind off-path reset. + clearCapture(); + const good_seq = peer.seq; + peer.seq = good_seq +% 5000; + var bogus = peer.segment(t.rst, 0, &.{}, false); + s.onFrame(bogus.bytes()); + try testing.expectEqual(ip.TcpState.established, s.tcpState()); + try testing.expectEqual(@as(usize, 1), cap_n); + try testing.expectEqual(t.ack_f, (try decode(sent(0))).flags); + + // The real thing. + peer.seq = good_seq; + var reset = peer.segment(t.rst, 0, &.{}, false); + s.onFrame(reset.bytes()); + try testing.expectEqual(ip.TcpState.closed, s.tcpState()); + try testing.expectError(error.ConnectionReset, s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqual(@as(u32, 2), s.counters.tcp_rst_rx); +} + +test "TCP: a segment for a different port is not mistaken for this connection" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + clearCapture(); + const real_port = peer.stack_port; + peer.stack_port = real_port ^ 1; + var stray = peer.segment(t.syn | t.ack_f, syn.seq +% 1, &.{}, true); + s.onFrame(stray.bytes()); + try testing.expectEqual(ip.TcpState.syn_sent, s.tcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "TCP: a segment from a different host is not mistaken for this connection" { + // The whole 4-tuple has to match, not just the ports. A stack that checks only the ports can + // have its connection completed - or reset - by any host on the segment that guesses a + // 16-bit number. + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + clearCapture(); + + // Same ports, different source address. + var impostor: Peer = .{ .ip = gw_ip, .port = 80, .mac = gw_mac, .seq = 0x7000_0000 }; + impostor.stack_port = peer.stack_port; + var stray = impostor.segment(t.syn | t.ack_f, syn.seq +% 1, &.{}, true); + s.onFrame(stray.bytes()); + try testing.expectEqual(ip.TcpState.syn_sent, s.tcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); + + // And a reset from the same impostor is ignored too. + var reset = impostor.segment(t.rst, 0, &.{}, false); + s.onFrame(reset.bytes()); + try testing.expectEqual(ip.TcpState.syn_sent, s.tcpState()); + try testing.expectEqual(@as(u32, 0), s.counters.tcp_rst_rx); +} + +test "TCP: the peer's MSS is honoured, and the request is split across segments" { + // The MSS option only matters when the request is bigger than it, which for a GET means a long + // path. A stack that ignores the option sends one oversized segment that a peer with a small + // MSS - a tunnel, a PPPoE link, anything with encapsulation overhead - drops silently. + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac, .mss = 100 }; + var out: [64]u8 = undefined; + const path: [300]u8 = @splat('q'); + var full_path: [301]u8 = undefined; + full_path[0] = '/'; + @memcpy(full_path[1..], &path); + + const syn = try startGet(&s, &peer, &full_path, &out); + const iss = syn.seq; + clearCapture(); + var synack = peer.segment(t.syn | t.ack_f, iss +% 1, &.{}, true); + s.onFrame(synack.bytes()); + peer.seq +%= 1; + + // Reassemble the request from however many segments it takes, acknowledging each one: with a + // window of one segment, nothing more is sent until the previous is acknowledged. + var assembled: [512]u8 = undefined; + var got: usize = 0; + var rounds: usize = 0; + while (true) : (rounds += 1) { + try testing.expect(rounds < 16); // termination, so a stall fails rather than hangs + try testing.expectEqual(@as(usize, 1), cap_n); + const seg = try decode(sent(0)); + try testing.expect(seg.data.len <= 100); // the peer's MSS, honoured + try testing.expectEqual(iss +% 1 +% @as(u32, @intCast(got)), seg.seq); + @memcpy(assembled[got..][0..seg.data.len], seg.data); + got += seg.data.len; + if (seg.flags & t.fin != 0) break; + clearCapture(); + var ack = peer.segment(t.ack_f, seg.seq +% @as(u32, @intCast(seg.data.len)), &.{}, false); + s.onFrame(ack.bytes()); + if (cap_n == 0) break; // request fully sent and acknowledged + } + try testing.expect(rounds >= 3); // 400-odd bytes at 100 per segment + try testing.expect(std.mem.startsWith(u8, assembled[0..got], "GET /qqq")); + try testing.expect(std.mem.endsWith(u8, assembled[0..got], "\r\n\r\n")); + try testing.expect(std.mem.indexOf(u8, assembled[0..got], &path) != null); +} + +test "TCP: sequence numbers wrap across 2^32 without stalling" { + var s = newStack(); + // A peer whose ISN is chosen so its data crosses the wrap. This is the case a `<` comparison + // instead of RFC 1982 serial arithmetic breaks, and it breaks by hanging forever. + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac, .seq = 0xffff_ffe0 }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + const iss = syn.seq; + const our_next = try handshake(&s, &peer, iss); + + const head = "HTTP/1.1 200 OK\r\nContent-Length: 8\r\n\r\n"; + clearCapture(); + var a = peer.segment(t.ack_f, our_next, head, false); // 38 bytes: crosses the wrap + s.onFrame(a.bytes()); + peer.seq +%= @intCast(head.len); + try testing.expect(peer.seq < 0x1000); // we really did wrap + + var b = peer.segment(t.ack_f, our_next, "12345678", false); + s.onFrame(b.bytes()); + try testing.expectEqual(@as(usize, 8), try s.httpGet(peer.ip, peer.port, "/", &out)); + try testing.expectEqualStrings("12345678", out[0..8]); +} + +test "TCP: an unresolvable peer fails with HostUnreachable after ARP gives up" { + var s = newStack(); + s.tick(0); + s.setStatic(our_ip, mask24, gw_ip); + var out: [64]u8 = undefined; + // Nothing in the cache, and nothing ever answers. + try testing.expectError(error.WouldBlock, s.httpGet(peer_ip, 80, "/", &out)); + try testing.expectEqual(ip.TcpState.arp_wait, s.tcpState()); + var now: u64 = 0; + var k: usize = 0; + while (k < 8) : (k += 1) { + now += 1000; + s.tick(now); + } + try testing.expectError(error.HostUnreachable, s.httpGet(peer_ip, 80, "/", &out)); + // Every attempt was a broadcast ARP request for the peer. + try testing.expect(s.counters.arp_tx >= 5); +} + +test "TCP: an off-net destination is sent to the gateway's MAC" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, gw_mac, gw_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + var out: [64]u8 = undefined; + try testing.expectError(error.WouldBlock, s.httpGet(off_net_ip, 80, "/", &out)); + try testing.expectEqual(@as(usize, 1), cap_n); + const syn = lastSent(); + try testing.expectEqualSlices(u8, &gw_mac, syn[0..6]); // to the gateway... + try testing.expectEqualSlices(u8, &off_net_ip, syn[14 + 16 ..][0..4]); // ...for the peer +} + +// ===================================================================================== HTTP + +/// Handshake, then feed the response in the given pieces, one segment each. +fn runResponse(s: *ip.Stack, peer: *Peer, path: []const u8, out: []u8, pieces: []const []const u8) !void { + const syn = try startGet(s, peer, path, out); + const iss = syn.seq; + const our_next = try handshake(s, peer, iss); + for (pieces) |piece| { + clearCapture(); + var seg = peer.segment(t.ack_f, our_next, piece, false); + s.onFrame(seg.bytes()); + peer.seq +%= @intCast(piece.len); + } +} + +test "HTTP: headers split across two segments" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + // The split falls inside the `Content-Length` field name, and the second piece carries the + // blank line and the start of the body. This is the ordinary case on a real server, and it is + // the one a parser that assumes headers arrive whole gets wrong. + try runResponse(&s, &peer, "/split", &out, &.{ + "HTTP/1.1 200 OK\r\nServer: nginx\r\nContent-Len", + "gth: 11\r\nETag: \"x\"\r\n\r\nhello wor", + "ld", + }); + const n = try s.httpGet(peer.ip, peer.port, "/split", &out); + try testing.expectEqual(@as(usize, 11), n); + try testing.expectEqualStrings("hello world", out[0..n]); + try testing.expectEqual(@as(u16, 200), s.httpStatus()); +} + +test "HTTP: the status line and blank line split one byte at a time" { + // The pathological segmentation: every byte its own segment. If any offset in the parser is + // off by one, one of these iterations lands on it. + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + const response = "HTTP/1.1 201 Created\r\nContent-Length: 3\r\nX: y\r\n\r\nabc"; + var pieces: [response.len][]const u8 = undefined; + for (&pieces, 0..) |*p, i| p.* = response[i .. i + 1]; + try runResponse(&s, &peer, "/bytes", &out, &pieces); + try testing.expectEqual(@as(usize, 3), try s.httpGet(peer.ip, peer.port, "/bytes", &out)); + try testing.expectEqualStrings("abc", out[0..3]); + try testing.expectEqual(@as(u16, 201), s.httpStatus()); +} + +test "HTTP: a header name's case is not significant" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + try runResponse(&s, &peer, "/case", &out, &.{ + "HTTP/1.0 200 OK\r\ncOnTeNt-LeNgTh: 7 \r\n\r\n1234567", + }); + try testing.expectEqual(@as(usize, 7), try s.httpGet(peer.ip, peer.port, "/case", &out)); + try testing.expectEqualStrings("1234567", out[0..7]); +} + +test "HTTP: a body with no Content-Length is terminated by the peer's FIN" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4096]u8 = undefined; + const syn = try startGet(&s, &peer, "/stream", &out); + const iss = syn.seq; + const our_next = try handshake(&s, &peer, iss); + + var a = peer.segment(t.ack_f, our_next, "HTTP/1.1 200 OK\r\nServer: x\r\n\r\npart one ", false); + s.onFrame(a.bytes()); + peer.seq +%= 39; + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, "/stream", &out)); + + var b = peer.segment(t.ack_f, our_next, "part two", false); + s.onFrame(b.bytes()); + peer.seq +%= 8; + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, "/stream", &out)); + + // RFC 7230 3.3.3 case 7: with no Content-Length and no chunking, the connection close is the + // framing. That is why the request said `Connection: close`. + clearCapture(); + var fin = peer.segment(t.fin | t.ack_f, our_next, &.{}, false); + s.onFrame(fin.bytes()); + const n = try s.httpGet(peer.ip, peer.port, "/stream", &out); + try testing.expectEqualStrings("part one part two", out[0..n]); + + // The peer closed first, so this is RFC 793's CLOSE-WAIT -> LAST-ACK: our FIN goes out + // acknowledging theirs, and the connection is not finished until that FIN is acknowledged. + try testing.expectEqual(@as(usize, 1), cap_n); + const ours = try decode(sent(0)); + try testing.expectEqual(t.fin | t.ack_f, ours.flags); + try testing.expectEqual(peer.seq +% 1, ours.ack); // their FIN consumed one sequence number + try testing.expectEqual(ip.TcpState.last_ack, s.tcpState()); + + // Their ACK of our FIN finishes it. + clearCapture(); + peer.seq +%= 1; + var final = peer.segment(t.ack_f, ours.seq +% 1, &.{}, false); + s.onFrame(final.bytes()); + try testing.expectEqual(ip.TcpState.time_wait, s.tcpState()); + try testing.expectEqual(@as(usize, 0), cap_n); // a bare ACK needs no answer + + // The peer's FIN again, because our ACK of it was lost. It has already been consumed, so it is + // "old" by one sequence number - and a stack that only accepts an exactly-in-order FIN answers + // nothing, leaving the peer retransmitting until it gives up and resets. + clearCapture(); + var again: Peer = peer; + again.seq = peer.seq -% 1; // the sequence number their FIN actually carried + var dup = again.segment(t.fin | t.ack_f, ours.seq +% 1, &.{}, false); + s.onFrame(dup.bytes()); + try testing.expectEqual(@as(usize, 1), cap_n); + const reack = try decode(sent(0)); + try testing.expectEqual(t.ack_f, reack.flags); + try testing.expectEqual(peer.seq, reack.ack); // still the sequence number past their FIN + try testing.expectEqual(ip.TcpState.time_wait, s.tcpState()); +} + +// ============================================================================= HTTP chunked +// +// RFC 7230 4.1. The framing is a size in hex, CRLF, that many bytes, CRLF, repeated, ended by a +// zero size, an optional trailer section and one more CRLF. Two things make it worth this many +// cases: the caller must see the decoded bytes and none of the framing, and a segment boundary +// may fall anywhere - including inside a size, inside a CRLF, and inside a chunk whose *data* +// contains CRLFs of its own. + +/// The example from RFC 7230's own appendix, by way of the one everybody quotes. Its third chunk +/// carries `\r\n\r\n` as data, which is the trap: a decoder that scans for a delimiter instead of +/// counting the size it was given loses the rest of the body here, and reports success. +const chunked_head = "HTTP/1.1 200 OK\r\nServer: cloudflare\r\nTransfer-Encoding: chunked\r\n\r\n"; +const chunked_wire = "4\r\nWiki\r\n5\r\npedia\r\nE\r\n in\r\n\r\nchunks.\r\n0\r\n\r\n"; +const chunked_want = "Wikipedia in\r\n\r\nchunks."; + +/// Drive a response through a fresh connection, cut into `pieces`, and return the decoded body. +fn decodeChunked(out: []u8, pieces: []const []const u8) ![]const u8 { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + try runResponse(&s, &peer, "/c", out, pieces); + const n = try s.httpGet(peer.ip, peer.port, "/c", out); + return out[0..n]; +} + +/// The same, expecting a named failure rather than a body. +fn expectChunkedError(want: anyerror, out: []u8, pieces: []const []const u8) !void { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + try runResponse(&s, &peer, "/c", out, pieces); + try testing.expectError(want, s.httpGet(peer.ip, peer.port, "/c", out)); +} + +test "HTTP chunked: a whole response in one segment decodes, framing bytes and all removed" { + var out: [256]u8 = undefined; + const got = try decodeChunked(&out, &.{chunked_head ++ chunked_wire}); + try testing.expectEqualStrings(chunked_want, got); + // Said the other way round, because it is the property that matters: no size, no CRLF and no + // terminator reached the caller. + try testing.expect(std.mem.indexOf(u8, got, "\r\nE\r\n") == null); + try testing.expect(std.mem.indexOf(u8, got, "0\r\n") == null); +} + +test "HTTP chunked: the response split at every single offset, two segments" { + // The decoder has to resume from wherever the cut landed: mid-size, between the CR and the LF + // of a chunk header, mid-data, mid-terminator. This walks every one of those positions. + const response = chunked_head ++ chunked_wire; + var split: usize = 1; + while (split < response.len) : (split += 1) { + var out: [256]u8 = undefined; + const got = try decodeChunked(&out, &.{ response[0..split], response[split..] }); + try testing.expectEqualStrings(chunked_want, got); + } +} + +test "HTTP chunked: the response split one byte at a time" { + // The pathological segmentation. Every state in the machine is entered with an empty input + // and re-entered with one byte, which is where a decoder that peeks at `b[1]` dies. + const response = chunked_head ++ chunked_wire; + var pieces: [response.len][]const u8 = undefined; + for (&pieces, 0..) |*p, i| p.* = response[i .. i + 1]; + var out: [256]u8 = undefined; + const got = try decodeChunked(&out, &pieces); + try testing.expectEqualStrings(chunked_want, got); +} + +test "HTTP chunked: the body arrives across three segments cut inside one chunk's data" { + var out: [256]u8 = undefined; + const got = try decodeChunked(&out, &.{ + chunked_head ++ "4\r\nWi", + "ki\r\n5\r\npe", + "dia\r\nE\r\n in\r\n\r\nchunks.\r\n0\r\n\r\n", + }); + try testing.expectEqualStrings(chunked_want, got); +} + +test "HTTP chunked: chunk extensions are skipped, not delivered" { + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n" ++ + "5;name=value;flag\r\nhello\r\n0;last\r\n\r\n", + }); + try testing.expectEqualStrings("hello", got); +} + +test "HTTP chunked: an extension split across segments is still skipped" { + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5;na", + "me=val", + "ue\r\nhello\r\n0\r\n\r\n", + }); + try testing.expectEqualStrings("hello", got); +} + +test "HTTP chunked: a trailer section is skipped and only its final CRLF completes the body" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + const syn = try startGet(&s, &peer, "/c", &out); + const our_next = try handshake(&s, &peer, syn.seq); + + // Everything up to but not including the CRLF that ends the trailer section. + const piece = + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n0\r\nExpires: now\r\n"; + var a = peer.segment(t.ack_f, our_next, piece, false); + s.onFrame(a.bytes()); + peer.seq +%= @intCast(piece.len); + + // The zero chunk is in and every body byte is here, and it is still not complete: the trailer + // section is part of the message, and a decoder that finished at the zero chunk would hand + // the caller a body while leaving the connection mid-message. + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, peer.port, "/c", &out)); + + var b = peer.segment(t.ack_f, our_next, "\r\n", false); + s.onFrame(b.bytes()); + peer.seq +%= 2; + try testing.expectEqual(@as(usize, 5), try s.httpGet(peer.ip, peer.port, "/c", &out)); + try testing.expectEqualStrings("hello", out[0..5]); +} + +test "HTTP chunked: sizes in upper case hex, and with leading zeros" { + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n" ++ + "00000A\r\n0123456789\r\nB\r\nabcdefghijk\r\n000\r\n\r\n", + }); + try testing.expectEqualStrings("0123456789abcdefghijk", got); +} + +test "HTTP chunked: an empty body is the terminator alone" { + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 204 No Content\r\nTransfer-Encoding: chunked\r\n\r\n0\r\n\r\n", + }); + try testing.expectEqual(@as(usize, 0), got.len); +} + +test "HTTP chunked: Content-Length beside chunked is ignored, not obeyed" { + // RFC 7230 3.3.3 case 3. A response carrying both is the request-smuggling disagreement, and + // the framing that wins is the chunked one. Obeying the length here would stop after 2 bytes + // and report success on a fifth of the body. + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nContent-Length: 2\r\nTransfer-Encoding: chunked\r\n\r\n" ++ + "5\r\nhello\r\n0\r\n\r\n", + }); + try testing.expectEqualStrings("hello", got); +} + +test "HTTP chunked: the header order does not decide which framing wins" { + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\nContent-Length: 2\r\n\r\n" ++ + "5\r\nhello\r\n0\r\n\r\n", + }); + try testing.expectEqualStrings("hello", got); +} + +test "HTTP chunked: a size with no hex digits is refused, never read as the terminator" { + // The dangerous misparse: a stray CRLF where a size belongs is a zero-length chunk to a + // decoder with no `1*HEXDIG` check, and a zero-length chunk ends the body. That is a + // truncated response reported as a complete one. + var out: [64]u8 = undefined; + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n\r\nhello\r\n0\r\n\r\n", + }); + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\nxyz\r\nhello\r\n0\r\n\r\n", + }); +} + +test "HTTP chunked: a chunk not followed by CRLF is refused" { + var out: [64]u8 = undefined; + // Data, then a bare LF where the CRLF belongs. + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\n0\r\n\r\n", + }); + // A chunk header whose CR is not followed by LF. + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\rhello\r\n0\r\n\r\n", + }); + // The final CRLF of the message, mangled. + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n0\r\n\rx", + }); +} + +test "HTTP chunked: each half of each CRLF is required in its own position" { + // The three cases above are all refused by a decoder that merely skips *two* bytes wherever a + // CRLF belongs; these are not. Each one is a well-framed message to such a decoder - it + // returns `hello` and reports success - and a malformed one to this stack. That is the + // difference between checking the delimiter and counting past it. + var out: [64]u8 = undefined; + // LF where the chunk's closing CR belongs, and the real LF behind it. + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\n\n0\r\n\r\n", + }); + // CR in place, then a byte that is not the LF. + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\rZ0\r\n\r\n", + }); + // And in the chunk header: CR in place, junk where the LF belongs. + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\rZhello\r\n0\r\n\r\n", + }); +} + +test "HTTP chunked: a second chunk with an empty size is refused, not read as the terminator" { + // The first chunk's size sets the "a digit was seen" flag, and it has to be cleared for the + // next one. Left set, the CRLF below reads as a zero-length chunk - the terminator - and the + // response ends silently five bytes in. + var out: [64]u8 = undefined; + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n\r\nmore\r\n0\r\n\r\n", + }); +} + +test "HTTP chunked: an impossible Content-Length beside chunked does not fail the request" { + // The other half of "chunked wins": the length is not merely unused for framing, it is not + // consulted at all - including by the check that refuses a body too big for `out`. A server + // that sends both is already not to be believed about the length. + var out: [64]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nContent-Length: 100000\r\nTransfer-Encoding: chunked\r\n\r\n" ++ + "5\r\nhello\r\n0\r\n\r\n", + }); + try testing.expectEqualStrings("hello", got); +} + +test "HTTP chunked: a body that exactly fills out still leaves window for its terminator" { + // The deadlock this pins: the advertised window is the room left in `out`, and chunked + // framing is consumed without going there. A body that fills `out` to the last byte closes + // the window, the terminator can never be accepted, and the request stalls against a peer + // that is behaving perfectly - until the RTO calls it a timeout. + var out: [5]u8 = undefined; + const got = try decodeChunked(&out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n", + "0\r\n\r\n", + }); + try testing.expectEqualStrings("hello", got); +} + +test "HTTP chunked: a size that overflows usize is refused, not wrapped" { + // Seventeen f's. Wrapped, this is a small number and the response looks well framed. + var out: [64]u8 = undefined; + try expectChunkedError(error.HttpChunkMalformed, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\nfffffffffffffffff\r\n", + }); +} + +test "HTTP chunked: a chunk larger than the caller's buffer fails on the header, before any copy" { + var out: [8]u8 = undefined; + try expectChunkedError(error.StreamTooLong, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n64\r\n", + }); +} + +test "HTTP chunked: chunks that together outgrow the buffer fail, and do not truncate" { + var out: [8]u8 = undefined; + try expectChunkedError(error.StreamTooLong, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n5\r\nworld\r\n0\r\n\r\n", + }); +} + +test "HTTP chunked: an endless chunk extension is bounded" { + const pad: [http_framing_over]u8 = @splat('x'); + var out: [4096]u8 = undefined; + try expectChunkedError(error.HttpHeadersTooLong, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5;", + &pad, + }); +} + +test "HTTP chunked: an endless trailer section is bounded" { + const pad: [http_framing_over]u8 = @splat('x'); + var out: [4096]u8 = undefined; + try expectChunkedError(error.HttpHeadersTooLong, &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n0\r\nX: ", + &pad, + }); +} + +/// One byte past the framing budget, so the bound is tested at the bound and not far above it. +const http_framing_over = ip.http_framing_max + 1; + +test "HTTP chunked: a close before the terminator is an error, not the body that did arrive" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + const syn = try startGet(&s, &peer, "/c", &out); + const our_next = try handshake(&s, &peer, syn.seq); + + const piece = "HTTP/1.1 200 OK\r\nTransfer-Encoding: chunked\r\n\r\n5\r\nhello\r\n"; + var a = peer.segment(t.ack_f, our_next, piece, false); + s.onFrame(a.bytes()); + peer.seq +%= @intCast(piece.len); + + var fin = peer.segment(t.fin | t.ack_f, our_next, &.{}, false); + s.onFrame(fin.bytes()); + // Five bytes of body are sitting in `out`, and they are not the answer: chunked framing says + // the message ends at the zero chunk, so a close before it truncated the response. + try testing.expectError(error.ConnectionClosed, s.httpGet(peer.ip, peer.port, "/c", &out)); +} + +test "HTTP: a transfer coding that is neither identity nor chunked is still refused" { + for ([_][]const u8{ "gzip", "deflate", "chunked, gzip", "gzip, chunked" }) |coding| { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + var head: [128]u8 = undefined; + const resp = try std.fmt.bufPrint( + &head, + "HTTP/1.1 200 OK\r\nTransfer-Encoding: {s}\r\n\r\n5\r\nhello\r\n0\r\n\r\n", + .{coding}, + ); + try runResponse(&s, &peer, "/tc", &out, &.{resp}); + try testing.expectError( + error.UnsupportedTransferEncoding, + s.httpGet(peer.ip, peer.port, "/tc", &out), + ); + try testing.expectEqual(ip.TcpState.closed, s.tcpState()); + } +} + +test "HTTP: Transfer-Encoding: identity is accepted" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + try runResponse(&s, &peer, "/id", &out, &.{ + "HTTP/1.1 200 OK\r\nTransfer-Encoding: identity\r\nContent-Length: 2\r\n\r\nok", + }); + try testing.expectEqual(@as(usize, 2), try s.httpGet(peer.ip, peer.port, "/id", &out)); +} + +test "HTTP: a malformed status line is refused" { + for ([_][]const u8{ + "ICY 200 OK\r\nContent-Length: 0\r\n\r\n", + "HTTP/1.1 200 OK\r\n\r\n", + "HTTP/1.1 2xx OK\r\n\r\n", + // The right shape, the wrong protocol. HTTP/2 has no textual status line at all, so a + // server answering this over a cleartext HTTP/1.1 request is not something to guess at. + "HTTP/2.0 200 OK\r\nContent-Length: 0\r\n\r\n", + "ICE/1.0 200 OK\r\nContent-Length: 0\r\n\r\n", + "HTTP/1.1\r\n\r\n", + }) |bad| { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + try runResponse(&s, &peer, "/bad", &out, &.{bad}); + try testing.expectError(error.HttpMalformed, s.httpGet(peer.ip, peer.port, "/bad", &out)); + } +} + +test "HTTP: a Content-Length larger than the caller's buffer fails before any body is copied" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [8]u8 = undefined; + try runResponse(&s, &peer, "/big", &out, &.{ + "HTTP/1.1 200 OK\r\nContent-Length: 100\r\n\r\n0123456789", + }); + try testing.expectError(error.StreamTooLong, s.httpGet(peer.ip, peer.port, "/big", &out)); +} + +test "HTTP: an impossible Content-Length fails at once, not after a partial body" { + // 100 promised bytes into an 8-byte buffer, and only five of them ever arrive. The request is + // already impossible when the headers are parsed, and saying so then is the difference between + // an immediate error and a request that hangs until the peer closes. + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [8]u8 = undefined; + try runResponse(&s, &peer, "/early", &out, &.{ + "HTTP/1.1 200 OK\r\nContent-Length: 100\r\n\r\n01234", + }); + try testing.expectError(error.StreamTooLong, s.httpGet(peer.ip, peer.port, "/early", &out)); + try testing.expectEqual(ip.TcpState.closed, s.tcpState()); +} + +test "HTTP: a body longer than the caller's buffer with no Content-Length fails" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [4]u8 = undefined; + try runResponse(&s, &peer, "/big2", &out, &.{ + "HTTP/1.1 200 OK\r\n\r\n0123456789", + }); + try testing.expectError(error.StreamTooLong, s.httpGet(peer.ip, peer.port, "/big2", &out)); +} + +test "HTTP: an oversized header block fails rather than truncating" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + // One header line per segment until the head buffer is full. No blank line ever arrives. + var pieces: [40][]const u8 = undefined; + for (&pieces) |*p| p.* = "X-Padding: 0123456789012345678901234567890123456789\r\n"; + var first: [2][]const u8 = .{ "HTTP/1.1 200 OK\r\n", pieces[0] }; + _ = &first; + try runResponse(&s, &peer, "/hdr", &out, &pieces); + try testing.expectError(error.HttpHeadersTooLong, s.httpGet(peer.ip, peer.port, "/hdr", &out)); +} + +test "HTTP: a Content-Length: 0 response completes on the headers alone" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + try runResponse(&s, &peer, "/empty", &out, &.{ + "HTTP/1.1 304 Not Modified\r\nContent-Length: 0\r\n\r\n", + }); + try testing.expectEqual(@as(usize, 0), try s.httpGet(peer.ip, peer.port, "/empty", &out)); + try testing.expectEqual(@as(u16, 304), s.httpStatus()); + // Completing the body half-closes, whatever the length was. + try testing.expect(s.tcpState() != .established); +} + +test "HTTP: a truncated body - FIN before Content-Length is met - is an error, not a short read" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + const syn = try startGet(&s, &peer, "/trunc", &out); + const iss = syn.seq; + const our_next = try handshake(&s, &peer, iss); + + const piece = "HTTP/1.1 200 OK\r\nContent-Length: 20\r\n\r\nshort"; + var a = peer.segment(t.ack_f, our_next, piece, false); + s.onFrame(a.bytes()); + peer.seq +%= @intCast(piece.len); + var fin = peer.segment(t.fin | t.ack_f, our_next, &.{}, false); + s.onFrame(fin.bytes()); + try testing.expectError(error.ConnectionClosed, s.httpGet(peer.ip, peer.port, "/trunc", &out)); +} + +test "HTTP: a non-default port appears in the Host header" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 8080, .mac = peer_mac }; + var out: [64]u8 = undefined; + const syn = try startGet(&s, &peer, "/", &out); + var synack = peer.segment(t.syn | t.ack_f, syn.seq +% 1, &.{}, true); + clearCapture(); + s.onFrame(synack.bytes()); + const req = try decode(sent(0)); + try testing.expect(std.mem.indexOf(u8, req.data, "\r\nHost: 192.168.1.90:8080\r\n") != null); +} + +// ========================================================================= the Host: header +// +// A name-based virtual host - which is what everything behind a CDN is - chooses the site from +// this header alone. `Host: 104.21.46.8` reaches Cloudflare and gets Cloudflare's error page; the +// site is only reachable by name. But a bare address in a lab is only reachable by address, so +// both spellings have to be exactly right. + +/// Start a request, complete the handshake, and return the request segment the stack sent. +fn requestFor(s: *ip.Stack, peer: *Peer, name: ?[]const u8, path: []const u8, out: []u8) !Seg { + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer.mac, peer.ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + try testing.expectError(error.WouldBlock, s.httpGetHost(peer.ip, name, peer.port, path, out)); + const syn = try decode(sent(0)); + peer.stack_port = syn.src_port; + clearCapture(); + var synack = peer.segment(t.syn | t.ack_f, syn.seq +% 1, &.{}, true); + s.onFrame(synack.bytes()); + return try decode(sent(0)); +} + +test "HTTP Host: a supplied name is sent instead of the address" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + const req = try requestFor(&s, &peer, "0x4200.cafe", "/", &out); + try testing.expect(std.mem.indexOf(u8, req.data, "\r\nHost: 0x4200.cafe\r\n") != null); + // The address is still where the connection went; the name is only ever a header. + try testing.expect(std.mem.indexOf(u8, req.data, "192.168.1.90") == null); +} + +test "HTTP Host: a name keeps the rule that only a non-default port is appended" { + var s80 = newStack(); + var peer80: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out80: [64]u8 = undefined; + const req80 = try requestFor(&s80, &peer80, "0x4200.cafe", "/", &out80); + try testing.expect(std.mem.indexOf(u8, req80.data, "\r\nHost: 0x4200.cafe\r\n") != null); + + var s8080 = newStack(); + var peer8080: Peer = .{ .ip = peer_ip, .port = 8080, .mac = peer_mac }; + var out8080: [64]u8 = undefined; + const req8080 = try requestFor(&s8080, &peer8080, "0x4200.cafe", "/", &out8080); + try testing.expect(std.mem.indexOf(u8, req8080.data, "\r\nHost: 0x4200.cafe:8080\r\n") != null); +} + +test "HTTP Host: no name is byte for byte what httpGet has always sent" { + // The working test against a bare address depends on this, so it is asserted on the bytes and + // not on a substring: two stacks with the same MAC and the same tick draw the same ephemeral + // port and the same ISN, so the two requests must be identical octet for octet. + var a = newStack(); + var peer_a: Peer = .{ .ip = peer_ip, .port = 8080, .mac = peer_mac }; + var out_a: [64]u8 = undefined; + const req_a = try requestFor(&a, &peer_a, null, "/index.html", &out_a); + var kept: [512]u8 = undefined; + @memcpy(kept[0..req_a.data.len], req_a.data); + const first = kept[0..req_a.data.len]; + + var b = newStack(); + var peer_b: Peer = .{ .ip = peer_ip, .port = 8080, .mac = peer_mac }; + var out_b: [64]u8 = undefined; + b.tick(1000); + b.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer_b.mac, peer_b.ip, zero_mac, our_ip, bcast_mac); + b.onFrame(probe.bytes()); + clearCapture(); + try testing.expectError(error.WouldBlock, b.httpGet(peer_b.ip, peer_b.port, "/index.html", &out_b)); + const syn = try decode(sent(0)); + peer_b.stack_port = syn.src_port; + clearCapture(); + var synack = peer_b.segment(t.syn | t.ack_f, syn.seq +% 1, &.{}, true); + b.onFrame(synack.bytes()); + const req_b = try decode(sent(0)); + + try testing.expectEqualSlices(u8, first, req_b.data); + try testing.expect(std.mem.indexOf(u8, req_b.data, "\r\nHost: 192.168.1.90:8080\r\n") != null); +} + +test "HTTP Host: the name is part of the request's identity, so changing it is Busy" { + var s = newStack(); + const peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var probe = arpFrame(1, peer.mac, peer.ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + + try testing.expectError(error.WouldBlock, s.httpGetHost(peer.ip, "0x4200.cafe", 80, "/", &out)); + // The same call again is the protocol. + try testing.expectError(error.WouldBlock, s.httpGetHost(peer.ip, "0x4200.cafe", 80, "/", &out)); + // A different virtual host on the same address for the same path is a different request, and + // riding on this connection would fetch the wrong site under the right name. + try testing.expectError(error.Busy, s.httpGetHost(peer.ip, "example.com", 80, "/", &out)); + // And "no name" is not the same request as any name. + try testing.expectError(error.Busy, s.httpGetHost(peer.ip, null, 80, "/", &out)); + try testing.expectError(error.Busy, s.httpGet(peer.ip, 80, "/", &out)); +} + +test "HTTP: httpGet before an address exists is refused" { + var s = newStack(); + var out: [64]u8 = undefined; + try testing.expectError(error.NoAddress, s.httpGet(peer_ip, 80, "/", &out)); +} + +test "HTTP: re-entering with different arguments is refused rather than silently switching" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + var other: [64]u8 = undefined; + _ = try startGet(&s, &peer, "/one", &out); + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, 80, "/one", &out)); + try testing.expectError(error.Busy, s.httpGet(peer.ip, 80, "/two", &out)); + try testing.expectError(error.Busy, s.httpGet(peer.ip, 81, "/one", &out)); + try testing.expectError(error.Busy, s.httpGet(gw_ip, 80, "/one", &out)); + // A different output buffer is the dangerous one: the body is written as it arrives, so the + // stack is holding a pointer into the first. + try testing.expectError(error.Busy, s.httpGet(peer.ip, 80, "/one", &other)); + // Same buffer, shorter: `Content-Length` was already checked against the original length, and + // the body is written through the original slice, so a shrunk view is just as wrong. + try testing.expectError(error.Busy, s.httpGet(peer.ip, 80, "/one", out[0..32])); + try testing.expectError(error.Busy, s.httpGet(peer.ip, 80, "/one", out[1..])); + // The original arguments still work. + try testing.expectError(error.WouldBlock, s.httpGet(peer.ip, 80, "/one", &out)); +} + +test "HTTP: a path longer than the request buffer is refused" { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + var out: [64]u8 = undefined; + const long: [600]u8 = @splat('a'); + try testing.expectError(error.RequestTooLong, s.httpGet(peer_ip, 80, &long, &out)); +} + +test "HTTP: two requests in sequence use different ephemeral ports" { + var s = newStack(); + var peer: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + var out: [64]u8 = undefined; + try runResponse(&s, &peer, "/a", &out, &.{"HTTP/1.1 200 OK\r\nContent-Length: 1\r\na\r\n\r\na"}); + _ = try s.httpGet(peer.ip, peer.port, "/a", &out); + const first_port = peer.stack_port; + + const peer2: Peer = .{ .ip = peer_ip, .port = 80, .mac = peer_mac }; + clearCapture(); + try testing.expectError(error.WouldBlock, s.httpGet(peer2.ip, peer2.port, "/b", &out)); + const syn = try decode(sent(0)); + try testing.expect(syn.src_port != first_port); +} + +// ====================================================================================== DNS +// +// RFC 1035. The header offsets and the name encoding below are written out again from the RFC, +// like every other wire format in this file. The parts that need testing are not the header - +// six 16-bit fields - but the two that are easy to get wrong and impossible to see when they are: +// matching the *question* as well as the id, and following compression pointers under a bound. + +/// RFC 1035 4.1.1, re-derived. +const q = struct { + const id = 0; + const flags = 2; + const qdcount = 4; + const ancount = 6; + const nscount = 8; + const arcount = 10; + const hlen = 12; +}; + +/// The resolver this network's DHCP server hands out: the gateway itself. +const dns_ip: ip.Ip4 = .{ 192, 168, 1, 1 }; + +/// RFC 1035 4.1.2 name encoding. No validation, deliberately: a test that shared the encoder's +/// checks could not write a malformed name to see the stack reject it. +fn wireName(buf: []u8, name: []const u8) usize { + var o: usize = 0; + var labels = std.mem.splitScalar(u8, name, '.'); + while (labels.next()) |label| { + buf[o] = @intCast(label.len); + @memcpy(buf[o + 1 ..][0..label.len], label); + o += 1 + label.len; + } + buf[o] = 0; + return o + 1; +} + +/// A DNS message under construction. +const Msg = struct { + buf: [512]u8 = @splat(0), + len: usize = 0, + + fn header(self: *Msg, id: u16, flags: u16, qd: u16, an: u16) void { + put16(&self.buf, q.id, id); + put16(&self.buf, q.flags, flags); + put16(&self.buf, q.qdcount, qd); + put16(&self.buf, q.ancount, an); + put16(&self.buf, q.nscount, 0); + put16(&self.buf, q.arcount, 0); + self.len = q.hlen; + } + + fn question(self: *Msg, name: []const u8, qtype: u16, qclass: u16) void { + self.len += wireName(self.buf[self.len..], name); + self.be(qtype); + self.be(qclass); + } + + /// Append one big-endian 16-bit field. + fn be(self: *Msg, v: u16) void { + put16(&self.buf, self.len, v); + self.len += 2; + } + + fn bytes(self: *Msg, b: []const u8) void { + @memcpy(self.buf[self.len..][0..b.len], b); + self.len += b.len; + } + + /// A resource record whose owner name is a compression pointer to `name_off`, which is what a + /// real server emits for every record after the first: the question's name is at offset 12, + /// and every answer points at it. + fn rr(self: *Msg, name_off: u16, rtype: u16, rclass: u16, rdata: []const u8) void { + self.be(0xc000 | name_off); + self.be(rtype); + self.be(rclass); + put32(&self.buf, self.len, 300); // TTL + self.len += 4; + self.be(@intCast(rdata.len)); + self.bytes(rdata); + } + + fn slice(self: *const Msg) []const u8 { + return self.buf[0..self.len]; + } +}; + +/// A UDP datagram from `src`:`sport` to our address at `dport`. +fn udpFrame(src: ip.Ip4, sport: u16, dport: u16, payload: []const u8) Frame { + var f: Frame = .{}; + f.eth(our_mac, gw_mac, 0x0800); + const seg_len = 8 + payload.len; + const p = f.ip4(src, our_ip, 17, seg_len); + put16(p, 0, sport); + put16(p, 2, dport); + put16(p, 4, @intCast(seg_len)); + put16(p, 6, 0); + @memcpy(p[8..], payload); + f.sealTransport(6); + return f; +} + +/// A stack with an address, a resolver, and the resolver's MAC already learnt. +fn newResolverStack() ip.Stack { + var s = newStack(); + s.tick(1000); + s.setStatic(our_ip, mask24, gw_ip); + s.setDnsServer(dns_ip); + var probe = arpFrame(1, gw_mac, dns_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + return s; +} + +/// The DNS payload of a captured query, with both checksums verified independently. Also returns +/// the source port, which is the other half of what an off-path spoofer has to guess. +fn queryOut(frame: []const u8) !struct { msg: []const u8, sport: u16 } { + try testing.expectEqual(@as(u16, 0x0800), be16(frame, 12)); + const h = frame[14..34]; + try testing.expectEqual(@as(u8, 17), h[9]); // UDP + try verify(h); + try testing.expectEqualSlices(u8, &dns_ip, h[16..20]); + const total = be16(h, 2); + const seg = frame[34 .. 14 + total]; + try testing.expectEqual(@as(u16, 53), be16(seg, 2)); + try testing.expectEqual(@as(u16, @intCast(seg.len)), be16(seg, 4)); + try verifyTransport(h[12..16].*, h[16..20].*, 17, seg); + return .{ .msg = seg[8..], .sport = be16(seg, 0) }; +} + +/// Answer the outstanding query with `an` answer records built by `fill`, and return the address +/// `resolve` then produces - or the error it produces. +fn answerWith(s: *ip.Stack, name: []const u8, m: *Msg) !ip.Ip4 { + var f = udpFrame(dns_ip, 53, dns_query_port, m.slice()); + s.onFrame(f.bytes()); + return s.resolve(name); +} + +/// The source port of the query most recently captured, filled in by `startResolve`. +var dns_query_port: u16 = 0; + +/// Start a query and record its id and source port. +fn startResolve(s: *ip.Stack, name: []const u8) !u16 { + try testing.expectError(error.WouldBlock, s.resolve(name)); + try testing.expectEqual(@as(usize, 1), cap_n); + const out = try queryOut(sent(0)); + dns_query_port = out.sport; + clearCapture(); + return be16(out.msg, q.id); +} + +test "DNS: the query is one A/IN question, recursion desired, from an ephemeral port" { + var s = newResolverStack(); + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); + try testing.expectEqual(@as(usize, 1), cap_n); + const out = try queryOut(sent(0)); + const msg = out.msg; + + try testing.expect(out.sport >= 49152); // RFC 6335 dynamic range + // QR=0, OPCODE=0, RD=1, and nothing else. RFC 1035 4.1.1. + try testing.expectEqual(@as(u16, 0x0100), be16(msg, q.flags)); + try testing.expectEqual(@as(u16, 1), be16(msg, q.qdcount)); + try testing.expectEqual(@as(u16, 0), be16(msg, q.ancount)); + try testing.expectEqual(@as(u16, 0), be16(msg, q.nscount)); + try testing.expectEqual(@as(u16, 0), be16(msg, q.arcount)); + + // The question: `6 0x4200 4 cafe 0`, then QTYPE=A, QCLASS=IN. Written out literally, because + // the length-prefixed encoding is the thing being checked. + const want = [_]u8{ 6, '0', 'x', '4', '2', '0', '0', 4, 'c', 'a', 'f', 'e', 0 }; + try testing.expectEqualSlices(u8, &want, msg[q.hlen..][0..want.len]); + try testing.expectEqual(@as(u16, 1), be16(msg, q.hlen + want.len)); // QTYPE=A + try testing.expectEqual(@as(u16, 1), be16(msg, q.hlen + want.len + 2)); // QCLASS=IN + try testing.expectEqual(@as(usize, q.hlen + want.len + 4), msg.len); + try testing.expectEqual(@as(u32, 1), s.counters.dns_tx); +} + +test "DNS: an answer resolves the name, and the query slot is released" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); // QR, RD, RA, RCODE 0 + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); + try testing.expectEqual(@as(u32, 1), s.counters.dns_rx); + // The slot is free again: a second name resolves without an intervening reset. + try testing.expectError(error.WouldBlock, s.resolve("example.com")); +} + +test "DNS: a CNAME ahead of the A record is stepped over, not read as an address" { + // This is the shape a CDN answers with, and a resolver that reads answer[0] gets a name where + // it wanted four octets. RDLENGTH would even be 4 for a short enough label. + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var cname: [32]u8 = undefined; + const cname_len = wireName(&cname, "edge.example"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 3); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 5, 1, cname[0..cname_len]); // CNAME + m.rr(q.hlen, 28, 1, &[_]u8{0} ** 16); // AAAA - also not an address this stack can use + m.rr(q.hlen, 1, 1, &[_]u8{ 172, 67, 221, 247 }); // and finally the A + + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 172, 67, 221, 247 }, &got); +} + +test "DNS: an owner name written out in full, not compressed, is skipped correctly" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + var full: [32]u8 = undefined; + m.bytes(full[0..wireName(&full, "0x4200.cafe")]); + m.be(1); // A + m.be(1); // IN + m.bytes(&[_]u8{ 0, 0, 1, 44 }); // TTL + m.be(4); + m.bytes(&[_]u8{ 104, 21, 46, 8 }); + + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); +} + +test "DNS: a compression pointer that loops is bounded, not followed forever" { + // The gadget: at the start of the answer section, a one-byte label followed by a pointer back + // to that label. Every jump goes strictly backwards - so the "pointers must point backwards" + // check that most parsers stop at passes it - and the walk still never ends, because stepping + // over the label moves forward again. Only counting the jumps terminates this. + // + // If this test hangs, it has failed. That is the whole point of it. + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + const gadget: u16 = @intCast(m.len); + m.bytes(&[_]u8{ 1, 'x' }); // a label... + m.be(0xc000 | gadget); // ...and a pointer back to it + + try testing.expectError(error.DnsMalformed, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: a compression pointer that points forward is rejected" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + // A forward pointer that a parser without the backwards rule would happily follow: it lands + // on a root label placed at the very end of this message, so the name resolves, the record + // behind it parses, and an address comes out. RFC 1035 4.1.4 only ever compresses against a + // *prior* occurrence, and the rule is what keeps `dnsSkipName`'s jumps monotone. + m.be(0xc000 | 0x002d); // -> offset 45, the root label appended below + m.be(1); // A + m.be(1); // IN + m.bytes(&[_]u8{ 0, 0, 1, 44 }); // TTL + m.be(4); + m.bytes(&[_]u8{ 6, 6, 6, 6 }); + try testing.expectEqual(@as(usize, 45), m.len); + m.bytes(&[_]u8{0}); // the root label the pointer aims at + try testing.expectError(error.DnsMalformed, answerWith(&s, "0x4200.cafe", &m)); + + // And one aimed past the end of the message entirely. + var s2 = newResolverStack(); + const id2 = try startResolve(&s2, "0x4200.cafe"); + var far: Msg = .{}; + far.header(id2, 0x8180, 1, 1); + far.question("0x4200.cafe", 1, 1); + far.be(0xc000 | 0x00fa); + try testing.expectError(error.DnsMalformed, answerWith(&s2, "0x4200.cafe", &far)); +} + +test "DNS: a reserved label type is refused rather than guessed past" { + // RFC 1035 4.1.4 defines the two top bits of a length byte: 00 is a label, 11 is a pointer, + // 01 and 10 are reserved. A parser that treats 0x40 as "a label of 64 bytes" walks somewhere + // arbitrary and then keeps going - here, straight onto a well-formed A record. + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.bytes(&[_]u8{0x40}); // reserved type, low bits zero + m.bytes(&([_]u8{'z'} ** 64)); // what a 0x40-as-length parser would skip + m.bytes(&[_]u8{0}); // ...landing on a root label, so the name "parses" + m.be(1); + m.be(1); + m.bytes(&[_]u8{ 0, 0, 1, 44 }); + m.be(4); + m.bytes(&[_]u8{ 6, 6, 6, 6 }); + try testing.expectError(error.DnsMalformed, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: a pointer to a self-referential offset in the question is bounded too" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + const here: u16 = @intCast(m.len); + // A pointer to itself: rejected by the backwards check alone, since the target is not less + // than the pointer's own offset. + m.be(0xc000 | here); + try testing.expectError(error.DnsMalformed, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: a response with the wrong transaction id is ignored, and the query stays live" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id +% 1, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 1, 2, 3, 4 }); + try testing.expectError(error.WouldBlock, answerWith(&s, "0x4200.cafe", &m)); + try testing.expectEqual(@as(u32, 0), s.counters.dns_rx); +} + +test "DNS: a response echoing a different question is ignored" { + // The id alone is 16 bits. A resolver that checks only the id accepts an answer for any name + // an attacker likes, which is the entire cache-poisoning family. + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("evil.example", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + try testing.expectError(error.WouldBlock, answerWith(&s, "0x4200.cafe", &m)); + + // The one that matters, and the one a length-blind check misses: a different name of exactly + // the same encoded length, so QTYPE and QCLASS still land where they are expected and every + // check but the name's own passes. `kafe` for `cafe`. + var lookalike: Msg = .{}; + lookalike.header(id, 0x8180, 1, 1); + lookalike.question("0x4200.kafe", 1, 1); + lookalike.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + // The same encoded length as the question we actually asked, so nothing after the name moves. + var ours: Msg = .{}; + ours.header(id, 0x8180, 1, 1); + ours.question("0x4200.cafe", 1, 1); + ours.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + try testing.expectEqual(ours.len, lookalike.len); + try testing.expectError(error.WouldBlock, answerWith(&s, "0x4200.cafe", &lookalike)); + + // Nor a right name asked as the wrong type or class. + var wrong_type: Msg = .{}; + wrong_type.header(id, 0x8180, 1, 1); + wrong_type.question("0x4200.cafe", 28, 1); // AAAA + wrong_type.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + try testing.expectError(error.WouldBlock, answerWith(&s, "0x4200.cafe", &wrong_type)); + + var wrong_class: Msg = .{}; + wrong_class.header(id, 0x8180, 1, 1); + wrong_class.question("0x4200.cafe", 1, 3); // CH + wrong_class.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + try testing.expectError(error.WouldBlock, answerWith(&s, "0x4200.cafe", &wrong_class)); +} + +test "DNS: the echoed question is matched case-insensitively, as RFC 4343 requires" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0X4200.CAFE", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); +} + +test "DNS: a response from the wrong source, or the wrong port, is ignored" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + + var wrong_src = udpFrame(.{ 192, 168, 1, 250 }, 53, dns_query_port, m.slice()); + s.onFrame(wrong_src.bytes()); + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); + + var wrong_port = udpFrame(dns_ip, 5353, dns_query_port, m.slice()); + s.onFrame(wrong_port.bytes()); + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); + + // And to a port that is not the one this query was sent from. + var wrong_dport = udpFrame(dns_ip, 53, dns_query_port +% 1, m.slice()); + s.onFrame(wrong_dport.bytes()); + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); + + // The right one still works, so the three rejections above are not rejecting everything. + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 6, 6, 6, 6 }, &got); +} + +test "DNS: a query, not a response, on the right port is ignored" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x0100, 1, 1); // QR clear + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 6, 6, 6, 6 }); + try testing.expectError(error.WouldBlock, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: NXDOMAIN and a refusal are distinct named errors" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var nx: Msg = .{}; + nx.header(id, 0x8183, 1, 0); // RCODE 3 + nx.question("0x4200.cafe", 1, 1); + try testing.expectError(error.NameNotFound, answerWith(&s, "0x4200.cafe", &nx)); + + var s2 = newResolverStack(); + const id2 = try startResolve(&s2, "0x4200.cafe"); + var refused: Msg = .{}; + refused.header(id2, 0x8185, 1, 0); // RCODE 5, REFUSED + refused.question("0x4200.cafe", 1, 1); + try testing.expectError(error.DnsRefused, answerWith(&s2, "0x4200.cafe", &refused)); +} + +test "DNS: an answer with no A record in it is NameNotFound, not a hang" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 28, 1, &[_]u8{0} ** 16); // AAAA only + try testing.expectError(error.NameNotFound, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: an A record with the wrong RDLENGTH is not read as an address" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 2); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 1, 2, 3 }); // an A record three bytes long + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); // the real one, behind it + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); +} + +test "DNS: an RDLENGTH that runs past the end of the message is refused, not read" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.be(0xc000 | q.hlen); + m.be(1); + m.be(1); + m.bytes(&[_]u8{ 0, 0, 1, 44 }); + m.be(400); // RDLENGTH far past what follows + m.bytes(&[_]u8{ 104, 21, 46, 8 }); + try testing.expectError(error.DnsMalformed, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: every truncation of a good response is refused, and none is read off the end" { + // Every prefix of a well-formed answer, each against a *fresh* query - which is the part that + // matters. Feeding them all to one query would stop testing after the first prefix that + // decided it, because a decided query stops listening, and the prefixes that cut inside the + // resource record - exactly the ones whose bounds are worth checking - come last. + var cut: usize = 0; + while (cut < 45) : (cut += 1) { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + try testing.expectEqual(@as(usize, 45), m.len); + + var f = udpFrame(dns_ip, 53, dns_query_port, m.buf[0..cut]); + s.onFrame(f.bytes()); + // Ignored or refused, but never resolved: a prefix of the truth is not the truth. + if (s.resolve("0x4200.cafe")) |_| return error.TestUnexpectedResult else |_| {} + } + // ...and the whole thing does resolve, so the loop above is rejecting truncation and not + // simply rejecting everything. + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); +} + +test "DNS: two queries in sequence use different source ports" { + // The id is 16 bits and the port is the other 16. Reusing one port halves what an off-path + // spoofer has to guess, and makes a late answer to the previous query land on the live one. + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + const first_port = dns_query_port; + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + _ = try answerWith(&s, "0x4200.cafe", &m); + + _ = try startResolve(&s, "example.com"); + try testing.expect(dns_query_port != first_port); +} + +test "DNS: an answer count larger than the answers present does not walk off the end" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 0xffff); // 65,535 answers promised, none delivered + m.question("0x4200.cafe", 1, 1); + try testing.expectError(error.DnsMalformed, answerWith(&s, "0x4200.cafe", &m)); +} + +test "DNS: the query is retransmitted on a doubling timer and then times out" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + + // Nothing before the first deadline. The query went out at t=1000 with a 1 s timer. + s.tick(1_999); + try testing.expectEqual(@as(usize, 0), cap_n); + + s.tick(2_000); + try testing.expectEqual(@as(usize, 1), cap_n); + const re = try queryOut(sent(0)); + // The same id, so an answer to the first attempt still counts. Redrawing it is how a slow + // resolver turns into a timeout on a network that was working. + try testing.expectEqual(id, be16(re.msg, q.id)); + clearCapture(); + + s.tick(3_999); + try testing.expectEqual(@as(usize, 0), cap_n); + s.tick(4_000); + try testing.expectEqual(@as(usize, 1), cap_n); + clearCapture(); + + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); + s.tick(8_000); + try testing.expectError(error.TimedOut, s.resolve("0x4200.cafe")); + try testing.expectEqual(@as(u32, 3), s.counters.dns_tx); + try testing.expectEqual(@as(u32, 2), s.counters.dns_retx); + + // And the slot is free: the next call starts a new query rather than returning the old error. + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); +} + +test "DNS: a late answer to an abandoned query does not resolve a new one" { + var s = newResolverStack(); + const first_id = try startResolve(&s, "0x4200.cafe"); + const first_port = dns_query_port; + // The whole schedule: 1 s, 2 s, 4 s, then out of tries. + s.tick(2_000); + s.tick(4_000); + s.tick(8_000); + clearCapture(); + try testing.expectError(error.TimedOut, s.resolve("0x4200.cafe")); + _ = try startResolve(&s, "0x4200.cafe"); + + var m: Msg = .{}; + m.header(first_id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 9, 9, 9, 9 }); + var f = udpFrame(dns_ip, 53, first_port, m.slice()); + s.onFrame(f.bytes()); + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); +} + +test "DNS: a second name while a query is in flight is Busy, and the first is untouched" { + var s = newResolverStack(); + const id = try startResolve(&s, "0x4200.cafe"); + try testing.expectError(error.Busy, s.resolve("example.com")); + // The same name, spelled with a trailing root dot and in a different case, is the same query. + try testing.expectError(error.WouldBlock, s.resolve("0X4200.CAFE.")); + try testing.expectError(error.WouldBlock, s.resolve("0x4200.cafe")); + + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + const got = try answerWith(&s, "0x4200.cafe.", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); +} + +test "DNS: with no resolver and no address, resolve says which one is missing" { + var no_server = newStack(); + no_server.tick(1000); + no_server.setStatic(our_ip, mask24, gw_ip); + clearCapture(); + try testing.expectError(error.NoDnsServer, no_server.resolve("0x4200.cafe")); + try testing.expectEqual(@as(usize, 0), cap_n); + + var no_addr = newStack(); + no_addr.tick(1000); + no_addr.setDnsServer(dns_ip); + clearCapture(); + try testing.expectError(error.NoAddress, no_addr.resolve("0x4200.cafe")); + try testing.expectEqual(@as(usize, 0), cap_n); +} + +test "DNS: an unusable name is refused before a byte leaves, and says which way it was unusable" { + var s = newResolverStack(); + const long: [ip.dns_name_max + 1]u8 = @splat('a'); + try testing.expectError(error.NameTooLong, s.resolve(&long)); + // A label over 63 bytes, inside a name that is itself short enough - so this is the label + // rule and not the name rule that rejects it. + const long_label = "b" ** 64; + for ([_][]const u8{ "", ".", "..", ".a", "a..b", long_label }) |bad| { + try testing.expectError(error.NameInvalid, s.resolve(bad)); + } + try testing.expectEqual(@as(usize, 0), cap_n); + // A name of exactly the maximum is fine, and is what proves the limit is off by nothing: + // 31 + 1 + 32 = 64 text bytes, encoding to 66 - which is `dns_qname_max` exactly. + const ok = "a" ** 31 ++ "." ++ "b" ** 32; + try testing.expectEqual(@as(usize, ip.dns_name_max), ok.len); + try testing.expectError(error.WouldBlock, s.resolve(ok)); +} + +test "DNS: the resolver DHCP supplied is the one resolve asks, with nothing configured" { + // The default path on this network: the lease carries option 6 and the caller does nothing. + var s = newStack(); + s.tick(10_000); + s.dhcpStart(); + const discover = try dhcpOut(sent(0)); + const xid = be32(discover, d.xid); + + var offer = dhcpReply(2, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(offer.bytes()); + var ack = dhcpReply(5, xid, our_ip, gw_ip, &standard_opts, our_ip, our_mac); + s.onFrame(ack.bytes()); + try testing.expectEqual(ip.DhcpState.bound, s.dhcpState()); + try testing.expectEqualSlices(u8, &dns_ip, &(s.dnsServer().?)); + + // The resolver's MAC, so the query can actually be addressed. + var probe = arpFrame(1, gw_mac, dns_ip, zero_mac, our_ip, bcast_mac); + s.onFrame(probe.bytes()); + clearCapture(); + + const id = try startResolve(&s, "0x4200.cafe"); + var m: Msg = .{}; + m.header(id, 0x8180, 1, 1); + m.question("0x4200.cafe", 1, 1); + m.rr(q.hlen, 1, 1, &[_]u8{ 104, 21, 46, 8 }); + const got = try answerWith(&s, "0x4200.cafe", &m); + try testing.expectEqualSlices(u8, &[_]u8{ 104, 21, 46, 8 }, &got); +} + +test "DNS: a new lease abandons a query in flight rather than leaving it to time out" { + var s = newResolverStack(); + _ = try startResolve(&s, "0x4200.cafe"); + s.dhcpStart(); + // No address and no resolver now, and the query is gone with them - so this is the error that + // names what is missing, not `Busy` from a query nobody can answer. + try testing.expectError(error.NoAddress, s.resolve("0x4200.cafe")); +} + +test "identity: the clock stirs the transaction ids, so two boots do not collide" { + // Same MAC, same firmware, different moment of first tick. If `tick` did not mix `now_ms` into + // the entropy, both would draw identical DHCP transaction ids and identical initial sequence + // numbers, and a reboot would happily accept a reply meant for its previous incarnation. + var a = newStack(); + a.tick(1234); + a.dhcpStart(); + const xid_a = be32(sent(0)[42..], d.xid); + + var b = newStack(); + b.tick(9_876_543); + b.dhcpStart(); + const xid_b = be32(sent(0)[42..], d.xid); + + try testing.expect(xid_a != xid_b); +} + +// ================================================================================ footprint + +test "footprint: the static cost of one Stack" { + // No printing. The test runner speaks a binary protocol over its own stdio under + // `zig build test`, and a diagnostic in the middle of it costs the whole suite's results for + // the sake of a number that an assertion states better anyway. + // + // 6 KiB is the ceiling, and it is not arbitrary: the image has ~128 KB of L2MEM, nothing + // initialises the 32 MB of PSRAM, and ESP-Hosted's queues and its task stacks compete for the + // same space. The stack is ~4,600 bytes today: 3,472 before chunked decoding and the resolver + // (104 bytes between them, mostly the encoded question), then 1,024 more when `http_head_max` + // went 1024 -> 2048 to fit a real CDN response head - measured at 1,043 bytes from the site this + // was pointed at, which failed the request by 19 bytes at the old size. + // + // A regression to 30 KiB would not announce itself any other way; it would show up as a stack + // overflow on the die. The heap in examples/http.zig was reduced by the same 2 KB this raise + // cost, so the image's total is unchanged. + const n = ip.Stack.footprint; + try testing.expect(n <= 6 * 1024); + // And a floor, so the ceiling cannot be met by quietly shrinking a buffer that the protocol + // needs: one full frame to build in, the request held for retransmission, the response head + // held while waiting for the blank line, and the DNS question held for the retransmissions + // and for the comparison against what the server echoes back. + try testing.expect(n >= ip.frame_max + ip.tcp_tx_max + ip.http_head_max + ip.dns_qname_max); +} diff --git a/src/net/libc.zig b/src/net/libc.zig new file mode 100644 index 0000000..38eab6d --- /dev/null +++ b/src/net/libc.zig @@ -0,0 +1,457 @@ +//! The libc symbols ESP-Hosted's C reaches for, and nothing more. +//! +//! This is not a libc. It is the exact set measured by linking the transport, and each entry is here +//! because a specific call site needs it: +//! +//! nm on the milestone-1 objects (transport_drv.o transport_util.o sdio_drv.o mempool.o) leaves +//! 26 undefined symbols. These are the libc ones: memcpy memset strcpy snprintf __errno_location +//! htole16 le16toh, plus malloc/free/realloc once mempool.c is included. +//! +//! Two sources cover them: +//! +//! 1. compiler_rt, which Zig links automatically. It provides the memory primitives - memcpy, +//! memset, memcmp, memmove - and the integer helpers clang emits for 64-bit division on a +//! 32-bit target, __udivdi3 and __divdi3. It provides no `str*` functions at all. +//! 2. This file, for everything else. +//! +//! The P4 mask ROM is a third possibility that this project deliberately does not use yet. +//! `components/esp_rom/esp32p4/ld/esp32p4.rom.newlib.ld` exports 32 newlib symbols - strlen, +//! strlcpy, strchr, strstr, memset, qsort, atoi and friends - as absolute addresses, which would +//! cost no code in the image and would be the same implementation IDF links. It is not wired in +//! because that file assigns those names unconditionally rather than with PROVIDE, so it would +//! collide with compiler_rt's own memset and memcpy definitions. Trading a real duplicate-symbol +//! hazard for a few hundred bytes is not worth it while the image is 2 KB. +//! +//! Deliberately absent: stdio beyond snprintf, locale, floating-point formatting beyond what +//! std.fmt gives, and anything reentrant. If a link error names a symbol not here, the honest move +//! is to add it here with a comment saying which call site wanted it - not to link a real libc. + +const std = @import("std"); + +/// Set by `install`. ESP-Hosted allocates per-packet buffers and frees them, so this cannot be an +/// arena; see the allocator discussion in src/net/port.zig. +var gpa: ?std.mem.Allocator = null; + +pub fn install(allocator: std.mem.Allocator) void { + gpa = allocator; +} + +// --------------------------------------------------------------------------------------------- +// malloc family +// +// C's `free` carries no size, but Zig's Allocator.free needs one. The classic fix is a header word +// in front of every block holding the length. It costs 8 bytes per allocation (the word plus +// padding to keep the payload 8-aligned, which the SDIO IDMAC path needs anyway) and it is the only +// way to bridge the two contracts without a side table. +// --------------------------------------------------------------------------------------------- + +/// Payload alignment. 8 rather than 4 because DMA descriptors on this chip want 8-byte alignment, +/// and buffers handed to CMD53 come from here. +const malloc_align: std.mem.Alignment = .@"8"; +const header_size = malloc_align.toByteUnits(); + +comptime { + // The header must not push the payload out of alignment. + std.debug.assert(header_size >= @sizeOf(usize)); + std.debug.assert(header_size % malloc_align.toByteUnits() == 0); +} + +fn allocBlock(total_payload: usize) ?[*]u8 { + const a = gpa orelse @panic("libc malloc before install()"); + const raw = a.rawAlloc(header_size + total_payload, malloc_align, @returnAddress()) orelse + return null; + // Record the payload length in the word directly before the payload. + const payload = raw + header_size; + @as(*usize, @ptrCast(@alignCast(raw))).* = total_payload; + return payload; +} + +fn payloadLen(payload: [*]u8) usize { + return @as(*const usize, @ptrCast(@alignCast(payload - header_size))).*; +} + +fn freeBlock(payload: [*]u8) void { + const a = gpa orelse @panic("libc free before install()"); + const len = payloadLen(payload); + a.rawFree((payload - header_size)[0 .. header_size + len], malloc_align, @returnAddress()); +} + +export fn malloc(size: usize) callconv(.c) ?*anyopaque { + if (size == 0) return null; + return @ptrCast(allocBlock(size)); +} + +export fn calloc(n: usize, size: usize) callconv(.c) ?*anyopaque { + const total = std.math.mul(usize, n, size) catch return null; + if (total == 0) return null; + const p = allocBlock(total) orelse return null; + @memset(p[0..total], 0); + return @ptrCast(p); +} + +export fn free(ptr: ?*anyopaque) callconv(.c) void { + const p = ptr orelse return; + freeBlock(@ptrCast(p)); +} + +export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque { + const p = ptr orelse return malloc(size); + if (size == 0) { + freeBlock(@ptrCast(p)); + return null; + } + const old: [*]u8 = @ptrCast(p); + const old_len = payloadLen(old); + if (old_len == size) return ptr; + + // Try to grow or shrink in place first; the allocator may well be able to, and mempool.c + // reallocs the same buffer repeatedly. + const a = gpa orelse @panic("libc realloc before install()"); + const whole = (old - header_size)[0 .. header_size + old_len]; + if (a.rawResize(whole, malloc_align, header_size + size, @returnAddress())) { + @as(*usize, @ptrCast(@alignCast(old - header_size))).* = size; + return ptr; + } + + const new = allocBlock(size) orelse return null; + @memcpy(new[0..@min(old_len, size)], old[0..@min(old_len, size)]); + freeBlock(old); + return @ptrCast(new); +} + +/// ESP-Hosted's `_h_malloc_align` path and IDF's `heap_caps_aligned_alloc` both land here. The +/// header trick still works as long as the requested alignment is not stricter than ours; anything +/// stricter would need the payload moved and the header written at a computed offset, and nothing +/// in the measured surface asks for that. Assert rather than silently misalign a DMA buffer. +export fn aligned_alloc(alignment: usize, size: usize) callconv(.c) ?*anyopaque { + // A stricter alignment would need the payload moved and the header written at a computed + // offset. Nothing in the measured surface asks for it, so this asserts rather than silently + // handing back a misaligned DMA buffer - which would corrupt a packet, not fail a call. + if (alignment > malloc_align.toByteUnits()) @panic("aligned_alloc: alignment stricter than 8"); + return malloc(size); +} + +// --------------------------------------------------------------------------------------------- +// string +// +// compiler_rt covers `mem*` and nothing else, so every `str*` ESP-Hosted references is here. The +// list is exactly what the link demanded - measured, not anticipated. +// --------------------------------------------------------------------------------------------- + +export fn strlen(s: [*:0]const u8) callconv(.c) usize { + // std.mem.len is the same loop; going through it keeps this honest about being a wrapper rather + // than a hand-optimised copy of something the standard library already has. + return std.mem.len(s); +} + +export fn strcpy(dst: [*]u8, src: [*:0]const u8) callconv(.c) [*]u8 { + var i: usize = 0; + while (src[i] != 0) : (i += 1) dst[i] = src[i]; + dst[i] = 0; + return dst; +} + +export fn strnlen(s: [*]const u8, max: usize) callconv(.c) usize { + var i: usize = 0; + while (i < max and s[i] != 0) : (i += 1) {} + return i; +} + +export fn strcmp(a: [*:0]const u8, b: [*:0]const u8) callconv(.c) c_int { + var i: usize = 0; + while (a[i] != 0 and a[i] == b[i]) : (i += 1) {} + return @as(c_int, a[i]) - @as(c_int, b[i]); +} + +export fn strncmp(a: [*]const u8, b: [*]const u8, n: usize) callconv(.c) c_int { + var i: usize = 0; + while (i < n) : (i += 1) { + if (a[i] != b[i]) return @as(c_int, a[i]) - @as(c_int, b[i]); + if (a[i] == 0) break; + } + return 0; +} + +// --------------------------------------------------------------------------------------------- +// endian helpers +// +// These are macros in musl's <endian.h>, but ESP-Hosted takes their address in a couple of places, +// so clang emits calls and the linker wants real symbols. riscv32 is little-endian, so both are +// identity - which is exactly why getting them wrong would be invisible here and corrupt on a +// big-endian host. Written as byte-order conversions rather than `return x` to say so. +// --------------------------------------------------------------------------------------------- + +export fn htole16(x: u16) callconv(.c) u16 { + return std.mem.nativeToLittle(u16, x); +} + +export fn le16toh(x: u16) callconv(.c) u16 { + return std.mem.littleToNative(u16, x); +} + +export fn htole32(x: u32) callconv(.c) u32 { + return std.mem.nativeToLittle(u32, x); +} + +export fn le32toh(x: u32) callconv(.c) u32 { + return std.mem.littleToNative(u32, x); +} + +// --------------------------------------------------------------------------------------------- +// errno +// +// ESP-Hosted reads errno after its own calls fail. There are no threads competing for it in a +// cooperative runtime, so one global is correct here; it would need to be per-task the moment a +// preemptive scheduler appeared. +// --------------------------------------------------------------------------------------------- + +var errno_storage: c_int = 0; + +export fn __errno_location() callconv(.c) *c_int { + return &errno_storage; +} + +// --------------------------------------------------------------------------------------------- +// snprintf +// +// The one genuinely non-trivial entry. ESP-Hosted uses it for log lines and for formatting MAC +// addresses and transport state, so the conversions that matter are %d %u %x %s %c %p and width / +// zero-pad on the integer ones. std.fmt does the formatting; this only parses the C format string. +// +// Unsupported conversions print `%!` followed by the specifier rather than being skipped, so a +// format this does not handle is visible in the log instead of silently dropping its argument. +// --------------------------------------------------------------------------------------------- + +export fn snprintf(buf: [*]u8, size: usize, fmt: [*:0]const u8, ...) callconv(.c) c_int { + var ap = @cVaStart(); + defer @cVaEnd(&ap); + return vsnprintfImpl(buf, size, fmt, &ap); +} + +export fn vsnprintf( + buf: [*]u8, + size: usize, + fmt: [*:0]const u8, + ap: *std.builtin.VaList, +) callconv(.c) c_int { + return vsnprintfImpl(buf, size, fmt, ap); +} + +/// `callconv(.c)` is required, not stylistic: `@cVaArg` is only available in a function using the C +/// calling convention, and Zig rejects it in an `auto` one. +fn vsnprintfImpl( + buf: [*]u8, + size: usize, + fmt: [*:0]const u8, + ap: *std.builtin.VaList, +) callconv(.c) c_int { + // Writes into the caller's buffer, tracking how many bytes *would* have been written, because + // that is what snprintf returns and callers use it to size a second call. + var out: Counting = .{ .buf = if (size == 0) &.{} else buf[0 .. size - 1] }; + + var i: usize = 0; + while (fmt[i] != 0) : (i += 1) { + if (fmt[i] != '%') { + out.byte(fmt[i]); + continue; + } + i += 1; + if (fmt[i] == '%') { + out.byte('%'); + continue; + } + + // flags and width: only the subset ESP-Hosted uses + var zero_pad = false; + var width: usize = 0; + while (fmt[i] == '0' or fmt[i] == '-' or fmt[i] == '+' or fmt[i] == ' ') : (i += 1) { + if (fmt[i] == '0') zero_pad = true; + } + while (fmt[i] >= '1' and fmt[i] <= '9') : (i += 1) { + width = width * 10 + (fmt[i] - '0'); + } + // length modifiers: consumed, and `ll`/`z` widen the fetch below + var long_long = false; + while (true) : (i += 1) { + switch (fmt[i]) { + 'l' => if (fmt[i + 1] == 'l') { + long_long = true; + } else {}, + 'h', 'z', 't', 'j' => {}, + else => break, + } + } + + switch (fmt[i]) { + 'd', 'i' => { + if (long_long) { + out.int(@cVaArg(ap, i64), 10, false, width, zero_pad); + } else { + out.int(@cVaArg(ap, c_int), 10, false, width, zero_pad); + } + }, + 'u' => { + if (long_long) { + out.int(@cVaArg(ap, u64), 10, false, width, zero_pad); + } else { + out.int(@cVaArg(ap, c_uint), 10, false, width, zero_pad); + } + }, + 'x' => out.int(@cVaArg(ap, c_uint), 16, false, width, zero_pad), + 'X' => out.int(@cVaArg(ap, c_uint), 16, true, width, zero_pad), + 'c' => out.byte(@truncate(@as(c_uint, @bitCast(@cVaArg(ap, c_int))))), + 's' => { + const s = @cVaArg(ap, ?[*:0]const u8) orelse "(null)"; + var n: usize = 0; + while (s[n] != 0) : (n += 1) {} + out.pad(width, n, ' '); + out.slice(s[0..n]); + }, + 'p' => { + out.slice("0x"); + out.int(@intFromPtr(@cVaArg(ap, ?*anyopaque)), 16, false, 8, true); + }, + 0 => break, + else => { + // Unsupported: say so in the output rather than desynchronising silently. The + // argument is deliberately not consumed - there is no way to know its width. + out.slice("%!"); + out.byte(fmt[i]); + }, + } + } + + if (size != 0) buf[@min(out.written, size - 1)] = 0; + return @intCast(out.would); +} + +/// A writer that stops filling at the end of the buffer but keeps counting, which is what +/// snprintf's return value means. +const Counting = struct { + buf: []u8, + written: usize = 0, + would: usize = 0, + + fn byte(self: *Counting, c: u8) void { + if (self.written < self.buf.len) { + self.buf[self.written] = c; + self.written += 1; + } + self.would += 1; + } + + fn slice(self: *Counting, s: []const u8) void { + for (s) |c| self.byte(c); + } + + fn pad(self: *Counting, width: usize, len: usize, fill: u8) void { + if (width > len) for (0..width - len) |_| self.byte(fill); + } + + fn int( + self: *Counting, + value: anytype, + base: u8, + upper: bool, + width: usize, + zero_pad: bool, + ) void { + var tmp: [24]u8 = undefined; + const end = std.fmt.printInt(&tmp, value, base, if (upper) .upper else .lower, .{}); + const s = tmp[0..end]; + self.pad(width, s.len, if (zero_pad) '0' else ' '); + self.slice(s); + } +}; + +// --------------------------------------------------------------------------------------------- +// Tests. These run on the host, where a wrong snprintf is cheap to find; on the die it would be a +// garbled log line at best and a buffer overrun at worst. +// --------------------------------------------------------------------------------------------- + +test "snprintf: the conversions esp_hosted actually uses" { + var buf: [64]u8 = undefined; + const n = snprintf(&buf, buf.len, "state %d port %u flags 0x%x %s", @as(c_int, -3), @as(c_uint, 7), @as(c_uint, 0xbeef), "ok"); + try std.testing.expectEqualStrings("state -3 port 7 flags 0xbeef ok", buf[0..@intCast(n)]); +} + +test "snprintf: return value is the length that would have been written" { + var buf: [8]u8 = undefined; + const n = snprintf(&buf, buf.len, "%s", "0123456789"); + // Truncated to 7 chars plus NUL, but reports the full 10 so a caller can size a second call. + try std.testing.expectEqual(@as(c_int, 10), n); + try std.testing.expectEqualStrings("0123456", buf[0..7]); + try std.testing.expectEqual(@as(u8, 0), buf[7]); +} + +test "snprintf: zero-padded width, as used for MAC bytes" { + var buf: [32]u8 = undefined; + const n = snprintf(&buf, buf.len, "%02x:%02x", @as(c_uint, 0x0a), @as(c_uint, 0xf1)); + try std.testing.expectEqualStrings("0a:f1", buf[0..@intCast(n)]); +} + +test "snprintf: size 0 writes nothing at all" { + var buf = [_]u8{0xAA} ** 4; + const n = snprintf(&buf, 0, "hello"); + try std.testing.expectEqual(@as(c_int, 5), n); + try std.testing.expectEqual(@as(u8, 0xAA), buf[0]); +} + +test "snprintf: an unsupported conversion is visible, not silent" { + var buf: [32]u8 = undefined; + const n = snprintf(&buf, buf.len, "f=%f", @as(f64, 1.5)); + try std.testing.expectEqualStrings("f=%!f", buf[0..@intCast(n)]); +} + +test "malloc/free/realloc survive the churn mempool.c generates" { + var backing: [4096]u8 = undefined; + var fba = std.heap.FixedBufferAllocator.init(&backing); + install(fba.allocator()); + defer gpa = null; + + // Same-size alloc/free churn: the case an arena cannot serve. + var i: usize = 0; + while (i < 8) : (i += 1) { + const p = malloc(64) orelse return error.OutOfMemory; + free(p); + } + + const a = malloc(32) orelse return error.OutOfMemory; + @memset(@as([*]u8, @ptrCast(a))[0..32], 0x5A); + const b = realloc(a, 64) orelse return error.OutOfMemory; + // Contents must survive the grow. + try std.testing.expectEqual(@as(u8, 0x5A), @as([*]u8, @ptrCast(b))[31]); + free(b); +} + +test "calloc zeroes, and rejects overflow rather than under-allocating" { + var backing: [1024]u8 = undefined; + var fba = std.heap.FixedBufferAllocator.init(&backing); + install(fba.allocator()); + defer gpa = null; + + const p = calloc(16, 4) orelse return error.OutOfMemory; + for (@as([*]u8, @ptrCast(p))[0..64]) |byte| try std.testing.expectEqual(@as(u8, 0), byte); + free(p); + + try std.testing.expect(calloc(std.math.maxInt(usize), 2) == null); +} + +test "strlen, strcpy, strcmp and strncmp agree with std" { + try std.testing.expectEqual(@as(usize, 0), strlen("")); + try std.testing.expectEqual(@as(usize, 3), strlen("abc")); + // strnlen stops at the bound, which is the whole reason the shim uses it on wire data. + try std.testing.expectEqual(@as(usize, 3), strnlen("abc", 8)); + try std.testing.expectEqual(@as(usize, 2), strnlen("abc", 2)); + try std.testing.expectEqual(@as(usize, 0), strnlen("abc", 0)); + + var dst: [8]u8 = undefined; + _ = strcpy(&dst, "abc"); + try std.testing.expectEqualStrings("abc", dst[0..3]); + try std.testing.expectEqual(@as(u8, 0), dst[3]); + + try std.testing.expect(strcmp("abc", "abc") == 0); + try std.testing.expect(strcmp("abc", "abd") < 0); + try std.testing.expect(strncmp("abcX", "abcY", 3) == 0); + try std.testing.expect(strncmp("abcX", "abcY", 4) != 0); +} diff --git a/src/net/link.zig b/src/net/link.zig new file mode 100644 index 0000000..4277595 --- /dev/null +++ b/src/net/link.zig @@ -0,0 +1,552 @@ +//! The seam: ESP-Hosted's station data channel, bridged to `src/net/ip.zig`. +//! +//! Everything below this file is proven - the SDIO host driver, the runtime, the port table, the +//! RPC layer, the association. Everything above it is proven too: `ip.zig` has 117 host tests and a +//! mutation sweep. This file is the twenty lines of pointer handling in between, and it is the one +//! part of the path that no host test can check, because both of its neighbours are C. +//! +//! So every decision here is cited rather than inferred. +//! +//! ------------------------------------------------------------------------------------------ +//! 1. Where the received frame starts: at `buffer`, offset zero. +//! +//! This is the single most expensive thing to get wrong. A frame shifted by the 12-byte +//! `esp_payload_header` parses as garbage - the ethertype lands in the middle of a MAC address - +//! and every one of ip.zig's tests would still pass. The RX convention is established by the +//! producer and confirmed by the vendor's own consumer: +//! +//! * sdio_drv.c:830 rejects any packet whose header `offset` field is not +//! `sizeof(struct esp_payload_header)`, so the payload always begins exactly one header in. +//! * sdio_drv.c:887 `buf_handle.payload = rxbuff + offset` - `payload` already points past the +//! header. `priv_buffer_handle` (:882) is what still points at the header. +//! * sdio_drv.c:1396-1400 allocates `copy_payload = _h_malloc(buf_handle->payload_len)` and +//! memcpy's `payload_len` bytes from `buf_handle->payload` into it, then frees the original +//! buffer at :1401. So the copy is exactly the payload, nothing more. +//! * sdio_drv.c:1407-1408 `rx(api_chan, copy_payload, copy_payload, payload_len)` - `buffer` +//! and `buff_to_free` are the same pointer, and it is the start of the frame. +//! * The vendor's own consumer agrees: esp_wifi_remote_net2.c:40-51 passes `buffer` straight to +//! the netif receive function as the frame and `buff_to_free` only as the free handle. +//! +//! `H_ESP_PAYLOAD_HEADER_OFFSET` appears on the *transmit* side only (transport_drv.c:381), where +//! ESP-Hosted is *building* a buffer and has to leave room for the header it is about to write. +//! Adding it on receive would be applying the same correction twice, in the wrong direction. +//! +//! ------------------------------------------------------------------------------------------ +//! 2. Who frees, and with what. +//! +//! `copy_payload` came from `_h_malloc` (sdio_drv.c:1396), so it is freed with `_h_free` - which +//! is exactly what `HOSTED_FREE` expands to (port_esp_hosted_host_os.h:139) and what +//! `transport_sta_free_cb` reaches through `MEMPOOL_FREE` with the pool disabled +//! (transport_util.h:29-31). `onRxFrame` below frees it through `g_h.funcs->_h_free`, once, on +//! every path including the error paths, and always returns `ESP_OK`. +//! +//! Returning `ESP_OK` unconditionally is not laziness, it is the only value that is safe under +//! both of sdio_drv.c's ownership rules. With `ESP_WIFI_REMOTE_VERSION` >= 1.3.1 the callee always +//! owns the buffer and the caller never frees (:1418). Below that, and when the macro is undefined, +//! the caller frees the buffer *if the callee returned non-zero* (:1411-1416). A non-zero return +//! from a callback that has already freed is therefore a double free under one rule and a leak +//! under neither - so this file frees and returns zero, which is one free under both. +//! +//! ------------------------------------------------------------------------------------------ +//! 3. `api_chan` must not be null. +//! +//! `transport_drv_sta_tx` opens with `assert(h && h == chan_arr[ESP_STA_IF]->api_chan)` +//! (transport_drv.c:369), and the vendor's reference RX callback opens with `assert(h)` +//! (esp_wifi_remote_net2.c:41). ESP-Hosted's own registration honours that: it allocates a cookie +//! and passes it in (esp_hosted_api.c:200-203). This build compiles the C at -O2 with `-DNDEBUG` +//! (Zig adds it for every non-Debug optimize mode), so those asserts are compiled out today and a +//! null cookie would merely be an unchecked contract violation rather than a crash - which is a +//! worse outcome, not a better one. `channel_cookie` below is that non-null cookie, and it is +//! handed back to `tx` on every transmit so the identity check holds. +//! +//! ------------------------------------------------------------------------------------------ +//! 4. The transmitted frame need not outlive the call. +//! +//! `transport_drv_sta_tx` allocates its own buffer and copies into it before queueing: +//! `mempool_alloc(..., MAX_TRANSPORT_BUFFER_SIZE, true)` at transport_drv.c:372 - with the pool +//! disabled that is `_h_malloc_align(1536, 64)` (transport_util.h:21-27) - then +//! `_h_memcpy(copy_buff + H_ESP_PAYLOAD_HEADER_OFFSET, buffer, len)` at :381, and only then +//! `esp_hosted_tx(..., copy_buff, ...)` at :383. Nothing retains `buffer`. That is what makes +//! `ip.Stack`'s "the slice is borrowed for the duration of the call" contract satisfiable, and it +//! is why `sendFrame` may hand over a pointer into the stack's single transmit staging buffer. +//! +//! ------------------------------------------------------------------------------------------ +//! 5. Why there is a re-entrancy guard. +//! +//! This is the one hazard the task description does not mention and it is real. +//! +//! `ip.Stack` is a single-threaded state machine: `onFrame` may send (an ARP reply, an ICMP echo +//! reply, a TCP ACK) before it returns, and `tick` and `httpGet` may too. Sending ends in +//! `esp_hosted_tx`, whose last act is +//! `_h_queue_item(to_slave_queue[prio], &buf_handle, HOSTED_BLOCK_MAX)` (sdio_drv.c:1607). That +//! queue holds four items (`CONFIG_ESP_HOSTED_SDIO_TX_Q_SIZE 4`, src/net/hosted/sdkconfig.h:41) +//! and `_h_queue_item` with `HOSTED_BLOCK_MAX` is a *blocking* send: port.zig:734-740 forwards it +//! to `os.Queue.send`, which suspends the calling task until there is room. +//! +//! So a full transmit queue suspends whoever is inside the stack. `onRxFrame` runs on ESP-Hosted's +//! `sdio_process_rx_task`; `tick` and `httpGet` run on the application's task. Without a guard, +//! either one can be suspended mid-mutation and the other walk straight into the same `Stack`. +//! On a cooperative scheduler that is not a torn read, it is two interleaved state machines +//! sharing one transmit buffer, one TCP sequence space and one `http.out` slice. +//! +//! The guard makes that impossible, and every way it can fire has a correct answer already: +//! +//! * a frame arriving while the stack is busy is dropped, which is what a real NIC does when its +//! transmit queue is full. DHCP, ARP and TCP all retransmit. +//! * a `tick` skipped is a `tick` deferred: `ip.zig`'s timers are absolute deadlines compared +//! against `now_ms` (`dhcpTick`, `tcpTick`), not increments, so nothing is lost. +//! * `httpGet` returns `error.WouldBlock`, which is precisely the answer its protocol already +//! requires the caller to handle by calling again with identical arguments. +//! +//! Each of those is counted, so a log can say which one happened rather than leaving a stall +//! unexplained. + +const std = @import("std"); + +const ip = @import("ip.zig"); +const port = @import("port.zig"); + +// ================================================================= ESP-Hosted's C surface + +/// `esp_hosted_if_type_t`, common/esp_hosted_interface.h:14-24. +/// +/// Note the value. The enumeration opens with `ESP_INVALID_IF`, so the station interface is **1**, +/// not 0. Registering channel 0 would fall through `transport_drv_add_channel`'s switch to +/// `default:` (transport_drv.c:481-484), which logs "Not yet supported" and returns NULL after +/// having already installed a half-built channel - and `chan_arr[ESP_STA_IF]` would stay NULL, so +/// sdio_drv.c:1394 would go on discarding every station frame in silence. +const esp_sta_if: c_uint = 1; + +/// `transport_channel_tx_fn_t`, transport_drv.h:118. Returns `esp_err_t`; 0 is `ESP_OK`. +const TxFn = *const fn (h: ?*anyopaque, buffer: ?*anyopaque, len: usize) callconv(.c) c_int; + +/// `transport_channel_rx_fn_t`, transport_drv.h:119. +const RxFn = *const fn ( + h: ?*anyopaque, + buffer: ?*anyopaque, + buff_to_free: ?*anyopaque, + len: usize, +) callconv(.c) c_int; + +/// transport_drv.h:134-136. `tx` is an out-parameter: the transport writes the interface's own +/// transmit function into it (transport_drv.c:469-471) and that is the only way to obtain it. +/// +/// This is compiled in - `transport_drv.c` is on build.zig's source list - but nothing calls it, +/// because the file that normally does (`esp_hosted_api.c`'s `add_esp_wifi_remote_channels`) is +/// not compiled: this project calls `setup_transport`, `rpc_init` and `transport_drv_reconfigure` +/// directly from `src/net/all.zig`. Registering the station channel is therefore ours to do. +extern fn transport_drv_add_channel( + api_chan: ?*anyopaque, + if_type: c_uint, + secure: u8, + tx: *?TxFn, + rx: RxFn, +) ?*anyopaque; + +/// The station's MAC, through the C shim (src/net/hosted/wifi_shim.c:96). It belongs to the C6's +/// radio, not to this chip, and ARP and Ethernet framing are built on it. Valid only after +/// `hosted_wifi_sta_start`, because that is what brings the radio up on the coprocessor. +extern fn hosted_wifi_get_mac(out: *[6]u8) c_int; + +// ============================================================================== module state + +/// The one IPv4 stack. A module-level variable rather than something the caller owns, because +/// `ip.Stack.send` is `*const fn ([]const u8) void` with no context pointer: the transmit callback +/// has to reach the transport some other way, and a file-scope binding is the honest version of +/// "some other way". 3,576 bytes of .bss - see `footprint`. +var sta: ip.Stack = undefined; + +/// The `api_chan` cookie. Its address is what ESP-Hosted stores and compares; its contents are +/// never read by anyone. See note 3 in the header for why it may not be null. +var channel_cookie: u32 = 0x5354_4100; // 'STA\0', so a memory dump names it + +/// The transport's station transmit function, from `transport_drv_add_channel`'s out-parameter. +var tx_fn: ?TxFn = null; + +/// Set once the channel is registered and the stack is live. +var opened: bool = false; + +/// The re-entrancy guard. See note 5 in the header. +var in_stack: bool = false; + +pub const Stats = struct { + /// Frames handed to us by sdio_drv.c, before any filtering. + rx_frames: u32 = 0, + /// Frames whose `h` was not our cookie. Non-zero means another channel's traffic reached this + /// callback, which would be an ESP-Hosted bug and not something to paper over. + rx_wrong_channel: u32 = 0, + /// `buffer` was null, or `len` was zero or larger than an Ethernet frame. + rx_bad: u32 = 0, + /// Frames dropped because the stack was already entered. See note 5. + rx_reentrant: u32 = 0, + /// Frames actually delivered to `ip.Stack.onFrame`. + rx_delivered: u32 = 0, + /// `tick` calls that found the stack entered and did nothing. + tick_skipped: u32 = 0, + /// `httpGet`/`httpGetHost` calls answered `WouldBlock` by the guard rather than by the stack. + http_deferred: u32 = 0, + /// `resolve` calls answered `WouldBlock` by the guard rather than by the stack. The query's + /// own timer runs in `tick`, so these cost a poll and never a retransmission. + dns_deferred: u32 = 0, + /// Frames handed to the transport. + tx_frames: u32 = 0, + /// Transmits the transport rejected: not ready, throttled, or out of buffers. + tx_failed: u32 = 0, + /// Transmits attempted before the channel existed. Should be zero. + tx_no_channel: u32 = 0, + /// Frames the stack asked to send, accepted into the deferred ring. The difference between this + /// and `tx_frames` is what is still waiting for the next `tick`. + tx_queued: u32 = 0, + /// Frames dropped because the deferred ring was full when the stack tried to send. Non-zero + /// means `tick` is not keeping up with the offered load; every protocol above this retransmits, + /// so it costs latency rather than correctness. + tx_ring_full: u32 = 0, + /// Frames the stack offered with an impossible length. Should be zero; a non-zero value points + /// at ip.zig rather than at the transport. + tx_bad: u32 = 0, +}; + +var counters: Stats = .{}; + +/// Everything this file adds to .bss, so the number in a report cannot rot. The stack dominates it. +pub const footprint: usize = + @sizeOf(@TypeOf(sta)) + + @sizeOf(@TypeOf(channel_cookie)) + + @sizeOf(@TypeOf(tx_fn)) + + @sizeOf(@TypeOf(opened)) + + @sizeOf(@TypeOf(in_stack)) + + @sizeOf(@TypeOf(counters)) + + @sizeOf(@TypeOf(tx_ring)); + +// ================================================================================= transmit + +/// `ip.Stack.send`. The slice is borrowed for the duration of this call only, which is exactly what +/// the transport needs - see note 4 in the header. +fn sendFrame(frame: []const u8) void { + if (tx_fn == null) { + counters.tx_no_channel += 1; + return; + } + if (frame.len == 0 or frame.len > ip.frame_max) { + counters.tx_bad += 1; + return; + } + // Queued, never transmitted from here. See `flushTx`. + const next = (tx_ring.head + 1) % tx_ring_slots; + if (next == tx_ring.tail) { + counters.tx_ring_full += 1; + return; + } + @memcpy(tx_ring.slot[tx_ring.head][0..frame.len], frame); + tx_ring.len[tx_ring.head] = @intCast(frame.len); + tx_ring.head = next; + counters.tx_queued += 1; +} + +/// Hand every queued frame to ESP-Hosted. MUST be called only from a task that may block. +/// +/// This indirection is the fix for a deadlock the board demonstrated, and it is worth stating +/// exactly because the shape of it is not obvious. +/// +/// `ip.Stack.onFrame` answers things: an ARP request gets a reply, an ICMP echo gets an echo, a TCP +/// segment gets an ACK. So a received frame turns into a transmitted frame inside `onFrame`. But +/// `onFrame` runs on ESP-Hosted's `sdio_process_rx_task`, and transmitting ends in +/// `_h_queue_item(to_slave_queue, HOSTED_BLOCK_MAX)` (sdio_drv.c:1607), which SUSPENDS the caller +/// when the queue is full. Suspend the RX task and it stops draining the receive queue; the receive +/// queue fills; ESP-Hosted logs "task still writing Rx data to queue!" and stops delivering. +/// Everything then looks like a dead IP stack. +/// +/// Measured on the board before this change: frames received froze at 17 and never advanced again, +/// no ping was ever answered, and the HTTP GET failed with HostUnreachable because the ARP reply it +/// needed was never sent. Raising the SDIO queue depth from 4 to 16 only moved the number. +/// +/// So the receive path now only ever copies into this ring, which cannot block, and the application +/// task drains it from `tick`. The cost is one copy and `tx_ring_slots * frame_max` of .bss. +fn flushTx() void { + const tx = tx_fn orelse return; + while (tx_ring.tail != tx_ring.head) { + const i = tx_ring.tail; + const n = tx_ring.len[i]; + counters.tx_frames += 1; + // The const cast is sound and it is load-bearing that it is: `transport_drv_sta_tx` reads + // `buffer` exactly once, as the source of a memcpy into its own aligned buffer + // (transport_drv.c:381), and neither writes through it nor retains it. ESP-Hosted's + // signature is simply not const-correct. + const rc = tx(@ptrCast(&channel_cookie), @ptrCast(&tx_ring.slot[i]), n); + if (rc != 0) counters.tx_failed += 1; + // Advance only after the call returns, so a frame is never handed out twice. + tx_ring.tail = (i + 1) % tx_ring_slots; + } +} + +/// Outgoing frames waiting for a task that may block. +/// +/// Four slots, at `ip.frame_max` each. Enough that the replies one pass of received frames can +/// generate - an ARP answer, an ICMP echo, a TCP ACK - all fit, since the whole ring is drained on +/// the very next `tick`. A full ring drops the newest frame and counts it, which is what a real +/// network interface does under load, and every protocol above this retransmits. +/// +/// Deliberately small: this is .bss competing with the heap ESP-Hosted allocates every received +/// frame from, and eight slots cost 12 KB that the transport needs more than this ring does. +const tx_ring_slots = 4; + +var tx_ring: struct { + slot: [tx_ring_slots][ip.frame_max]u8 = undefined, + len: [tx_ring_slots]u16 = @splat(0), + head: usize = 0, + tail: usize = 0, +} = .{}; + +// ================================================================================== receive + +/// `transport_channel_rx_fn_t`. Called from ESP-Hosted's `sdio_process_rx_task` +/// (sdio_drv.c:1407), which is one of the tasks `port.zig` spawned on this project's own runtime. +/// +/// The buffer is ours the moment this is entered, and it is freed on every path. See notes 1 and 2. +fn onRxFrame( + h: ?*anyopaque, + buffer: ?*anyopaque, + buff_to_free: ?*anyopaque, + len: usize, +) callconv(.c) c_int { + // `HOSTED_FREE(buff)` is `g_h.funcs->_h_free(buff)` (port_esp_hosted_host_os.h:139), and this + // is that call. First statement in the function so that no early return can miss it: the + // failure mode of a missed free here is not a leak that shows up in a heap report, it is the + // 32 KiB heap exhausted in a few seconds of the AP's broadcast traffic. + defer port.g_h.funcs.free(buff_to_free); + + counters.rx_frames += 1; + + if (h != @as(?*anyopaque, @ptrCast(&channel_cookie))) { + counters.rx_wrong_channel += 1; + return 0; + } + const bytes: [*]const u8 = @ptrCast(buffer orelse { + counters.rx_bad += 1; + return 0; + }); + if (!opened or len == 0 or len > ip.frame_max) { + counters.rx_bad += 1; + return 0; + } + if (in_stack) { + counters.rx_reentrant += 1; + return 0; + } + + in_stack = true; + defer in_stack = false; + counters.rx_delivered += 1; + sta.onFrame(bytes[0..len]); + return 0; +} + +// ================================================================================ lifecycle + +pub const Error = error{ + /// `hosted_wifi_get_mac` failed, or answered with the all-zero MAC that means "no radio yet". + /// The usual cause is calling this before `hosted_wifi_sta_start`. + MacUnavailable, + /// `transport_drv_add_channel` refused, or accepted without filling in the transmit function. + ChannelRegisterFailed, + AlreadyOpen, +}; + +/// Register the station channel and bring the IP stack up behind it. +/// +/// Call after `net.init` and after `hosted_wifi_sta_start`; association may follow or may already +/// have happened, it makes no difference to this. Registering *before* associating is the tidier +/// order, because `chan_arr[ESP_STA_IF]` becoming non-null is the moment sdio_drv.c stops +/// discarding station frames, and until then a live association fills ESP-Hosted's receive queue +/// and logs "task still writing Rx data to queue!". +/// +/// The order inside matters: the stack is constructed *before* the channel is registered. The +/// instant `transport_drv_add_channel` returns, `sdio_process_rx_task` may call `onRxFrame`, and +/// that must not find `sta` uninitialised. +pub fn open() Error!void { + if (opened) return error.AlreadyOpen; + + var mac_bytes: [6]u8 = @splat(0); + if (hosted_wifi_get_mac(&mac_bytes) != 0) return error.MacUnavailable; + // An all-zero MAC is not a MAC. It is what the shim hands back if the coprocessor answered + // without having a station interface, and building an ARP cache on it would produce a stack + // that transmits frames no switch will ever route back. + if (std.mem.allEqual(u8, &mac_bytes, 0)) return error.MacUnavailable; + + sta = .init(mac_bytes, &sendFrame); + + var tx: ?TxFn = null; + const channel = transport_drv_add_channel( + @ptrCast(&channel_cookie), + esp_sta_if, + 0, // secure=0: plain text, as ESP-Hosted itself uses for the two Wi-Fi interfaces + // (esp_hosted_api.c:105-107). The secure path is the RPC channel's, and RPC has + // its own already. + &tx, + &onRxFrame, + ); + if (channel == null) return error.ChannelRegisterFailed; + // Belt and braces: the switch at transport_drv.c:467-485 is the only writer of `*tx`, and the + // one branch that leaves it untouched also returns NULL. Checking both means a future + // ESP-Hosted that separates those cannot leave us with a live channel and no way to transmit. + tx_fn = tx orelse return error.ChannelRegisterFailed; + + opened = true; +} + +/// True once `open` has succeeded. +pub fn isOpen() bool { + return opened; +} + +// ============================================================ the guarded entry points +// +// Every function that can mutate the stack goes through `in_stack`. Every function that only reads +// it does not, because a read cannot suspend and the worst it can observe is a value one frame out +// of date. + +/// Advance the stack's clock. Returns false if the stack was busy and the tick was skipped, which +/// is harmless - see note 5 - but worth being able to see. +pub fn tick(now_ms: u64) bool { + if (in_stack) { + counters.tick_skipped += 1; + return false; + } + in_stack = true; + sta.tick(now_ms); + in_stack = false; + // Outside the guard, and last: draining may block, and `in_stack` must not be held across a + // suspension or the receive path would drop every frame that arrived while we waited. + flushTx(); + return true; +} + +/// Begin DHCP. Call `tick` at least once first: `dhcpStart` stamps the acquisition's start time +/// from the stack's idea of now, which only `tick` sets. Returns false if the stack was busy. +pub fn dhcpStart() bool { + if (in_stack) return false; + in_stack = true; + defer in_stack = false; + sta.dhcpStart(); + return true; +} + +/// Configure statically instead of asking a server. +pub fn setStatic(addr: [4]u8, mask: [4]u8, gw: [4]u8) bool { + if (in_stack) return false; + in_stack = true; + defer in_stack = false; + sta.setStatic(addr, mask, gw); + return true; +} + +/// Override the resolver `resolve` asks. Not needed on a network whose DHCP server offers one - +/// `dhcpBind` stores option 6 and `resolve` uses it with no configuration at all. Returns false if +/// the stack was busy. +pub fn setDnsServer(addr: [4]u8) bool { + if (in_stack) return false; + in_stack = true; + defer in_stack = false; + sta.setDnsServer(addr); + return true; +} + +/// One HTTP GET, with the address literal as the `Host:` header. `ip.Stack.httpGet`'s protocol, +/// unchanged: this returns `error.WouldBlock` until the body is complete, and the caller must keep +/// calling with *identical* arguments while driving `tick`. `out` is borrowed until a length comes +/// back. +pub fn httpGet(host: [4]u8, remote_port: u16, path: []const u8, out: []u8) ip.HttpError!usize { + return httpGetHost(host, null, remote_port, path, out); +} + +/// The same, with an explicit `Host:` name for a name-based virtual host. See +/// `ip.Stack.httpGetHost`; `name` is part of the request's identity, so it must not change between +/// calls any more than `path` may. +pub fn httpGetHost( + host: [4]u8, + name: ?[]const u8, + remote_port: u16, + path: []const u8, + out: []u8, +) ip.HttpError!usize { + if (in_stack) { + // Answering the caller's own protocol back at it. The alternative - waiting - would be a + // second place in this file that can block, and the guard exists to have exactly none. + counters.http_deferred += 1; + return error.WouldBlock; + } + in_stack = true; + defer in_stack = false; + return sta.httpGetHost(host, name, remote_port, path, out); +} + +/// Resolve a name to an address. `ip.Stack.resolve`'s protocol, which is `httpGet`'s: this returns +/// `error.WouldBlock` until an address or a real error comes back, and the caller keeps calling +/// with the same name while driving `tick`. +/// +/// The guard's answer is the same `error.WouldBlock`, for the same reason it is in `httpGetHost`: +/// the query's own retransmissions run in `sta.tick`, so a deferred poll costs nothing and the 7 s +/// bound still holds. Frames the query sends go through `sendFrame` into the deferred ring like +/// every other frame here - nothing on this path touches the transport's tx function directly. +pub fn resolve(name: []const u8) ip.DnsError!ip.Ip4 { + if (in_stack) { + counters.dns_deferred += 1; + return error.WouldBlock; + } + in_stack = true; + defer in_stack = false; + return sta.resolve(name); +} + +// ==================================================================== read-only accessors + +/// The station MAC the stack was built on. +pub fn mac() [6]u8 { + return sta.mac; +} + +/// The configured address, or null if there is none yet. +pub fn address() ?[4]u8 { + return sta.addr; +} + +pub fn netmask() [4]u8 { + return sta.mask; +} + +pub fn gateway() [4]u8 { + return sta.gw; +} + +pub fn dnsServer() ?[4]u8 { + return sta.dns; +} + +pub fn dhcpState() ip.DhcpState { + return sta.dhcp.state; +} + +pub fn tcpState() ip.TcpState { + return sta.tcp.state; +} + +pub fn httpStatus() u16 { + return sta.http.status; +} + +/// The IP stack's own counters: frames in, frames dropped, echoes answered, checksums rejected. +pub fn ipCounters() ip.Counters { + return sta.counters; +} + +/// This file's counters: the transport boundary, and every way the guard fired. +pub fn stats() Stats { + return counters; +} + +// There are no tests here, and that is an answer rather than an omission. Two of the three things +// this file does are calls into ESP-Hosted's C - `transport_drv_add_channel` and the transmit +// function it hands back - and the third is a callback that C invokes. A host test could only +// exercise it against a mock of the very code whose conventions are the thing in doubt, and it +// would pass just as happily against a mock that put the frame one header too late. The evidence +// that matters is the citations in this file's header and a board that answers a ping. diff --git a/src/net/port.zig b/src/net/port.zig new file mode 100644 index 0000000..fafbf4e --- /dev/null +++ b/src/net/port.zig @@ -0,0 +1,2194 @@ +//! ESP-Hosted's `g_h.funcs` port table, in Zig. +//! +//! This is the seam. Above it sit ~13,000 lines of ESP-Hosted C - the SDIO transport state machine, +//! the RPC protocol, the protobuf codec - which are already correct and which this project has no +//! intention of rewriting. Below it sit `std.Io`, `std.mem.Allocator` and `src/hal`. Everything +//! ESP-Hosted asks of an operating system passes through the 71 function pointers defined here, so +//! this file is the entire dependency of that C on FreeRTOS and ESP-IDF, and replacing it replaces +//! both. +//! +//! # The struct, and why its layout is the dangerous part +//! +//! `hosted_osi_funcs_t` is declared at `host/esp_hosted_os_abstraction.h:13-117`. Every member is a +//! function pointer, so on rv32 the struct is 71 words and **there is nothing in the type system, +//! on either side, that notices a field in the wrong place**. A mis-ordered pointer is a call to +//! the wrong function with the wrong arguments, which on this board is a hang with no console +//! output. +//! +//! Worse, the C struct is not one layout. Four mempool members are guarded by +//! `#ifdef H_USE_MEMPOOL` (`:64-69`), and `H_USE_MEMPOOL` is *always defined* - to 1 or to 0 - by +//! `host/port/esp/freertos/include/port_esp_hosted_host_config.h:127-131`, which `#ifdef` does not +//! care about. A translation unit that reaches the struct without having seen that header first +//! gets a struct 16 bytes shorter, with everything from `_h_config_gpio` onward displaced by four +//! pointers. That is reachable in the real tree: `host/esp_hosted.h:14` and +//! `host/drivers/transport/transport_util.h:10` both include the abstraction header as their first +//! include. Measured with our own flags: +//! +//! without -include port_esp_hosted_host_config.h: sizeof=268 _h_config_gpio=132 _h_event_post=264 +//! with -include port_esp_hosted_host_config.h: sizeof=284 _h_config_gpio=148 _h_event_post=280 +//! +//! This file targets the long layout, and `layout_check` below asserts the three numbers on the +//! right. The build force-includes that header into every ESP-Hosted translation unit and compares +//! C's `offsetof` against these assertions, so an include-order change fails the build instead of +//! the board. +//! +//! # What is real, what is a loud stub +//! +//! Real: memory, sync, threads, timers, time, GPIO, SDIO, events, mempool locks. That is every +//! entry the SDIO transport and the RPC layer touch, established by grepping the tree for each +//! `_h_` name rather than by guessing. +//! +//! Loud stubs: the SPI, SPI-HD and UART transports (a different bus), power-save (needs +//! `esp_sleep`), `_h_do_bus_transfer` (SPI-only; ESP-IDF leaves it null under SDIO), and +//! `_h_printf`. Each prints its own name through `ets_printf` and returns a failure code, so an +//! unimplemented path announces itself on the console instead of jumping through a null pointer. +//! `stub_calls` counts them. +//! +//! # Where ESP-Hosted's assumptions do not fit a cooperative single-core runtime +//! +//! Four places, all documented at the point of impact: +//! +//! * `_h_post_semaphore_from_isr` - FreeRTOS manipulates the semaphore inside a critical section +//! and requests a context switch on return. See `hosted_os.Semaphore.postFromIsr`. +//! * `_h_thread_cancel` - `vTaskDelete` kills a task where it stands; `std.Io`'s cancel asks and +//! waits, and ESP-Hosted's task bodies never return. See `hosted_os.Thread.cancel`. +//! * `_h_blocking_delay` - a deliberate busy-wait, which on a cooperative scheduler starves +//! every other task for its duration. Unused in the tree; kept honest. +//! * bounded waits - `std.Io` has no timed acquire for a mutex, semaphore or queue, so those +//! poll. See `hosted_os.poll_interval_ms`. Unbounded waits, which is what every hot path uses, +//! block properly. + +const std = @import("std"); +const assert = std.debug.assert; +const Io = std.Io; +const Allocator = std.mem.Allocator; + +const hal = @import("hal"); +const hheap = @import("heap.zig"); +const os = @import("hosted_os.zig"); + +const ret = os.ret; + +/// `ets_printf` from the mask ROM. Declared here rather than imported from `soc` so this file's +/// only module dependency is `hal`; the symbol comes from +/// `components/esp_rom/esp32p4/ld/esp32p4.rom.ld`. +extern fn ets_printf(fmt: [*:0]const u8, ...) c_int; + +fn note(comptime fmt: [*:0]const u8, args: anytype) void { + _ = @call(.auto, ets_printf, .{fmt} ++ args); +} + +// ============================================================================ the struct + +/// `void (*start_routine)(void const *)`, `esp_hosted_os_abstraction.h:25`. +pub const StartRoutine = *const fn (?*const anyopaque) callconv(.c) void; +/// `void (*timeout_handler)(void *)`, `:60`. +pub const TimerHandler = *const fn (?*anyopaque) callconv(.c) void; +/// `void (*gpio_isr_handler)(void* arg)`, `:73`. +pub const IsrHandler = *const fn (?*anyopaque) callconv(.c) void; +/// `esp_event_base_t`, which is `const char *`. +pub const EventBase = [*:0]const u8; + +/// `hosted_osi_funcs_t`, `host/esp_hosted_os_abstraction.h:13-117`, in the layout that +/// `H_USE_MEMPOOL` being defined produces. Field order is the C declaration order exactly; the +/// line number beside each is its declaration in that header. +pub const HostedOsiFuncs = extern struct { + // ---- Memory, :15-22 + /// :15 `void* (*)(void* dest, const void* src, uint32_t size)` + memcpy: *const fn (?*anyopaque, ?*const anyopaque, u32) callconv(.c) ?*anyopaque, + /// :16 `void* (*)(void* buf, int val, size_t len)` + memset: *const fn (?*anyopaque, c_int, usize) callconv(.c) ?*anyopaque, + /// :17 `void* (*)(size_t size)` + malloc: *const fn (usize) callconv(.c) ?*anyopaque, + /// :18 `void* (*)(size_t blk_no, size_t size)` + calloc: *const fn (usize, usize) callconv(.c) ?*anyopaque, + /// :19 `void (*)(void* ptr)` + free: *const fn (?*anyopaque) callconv(.c) void, + /// :20 `void* (*)(void *mem, size_t newsize)` + realloc: *const fn (?*anyopaque, usize) callconv(.c) ?*anyopaque, + /// :21 `void* (*)(size_t size, size_t align)` + malloc_align: *const fn (usize, usize) callconv(.c) ?*anyopaque, + /// :22 `void (*)(void* ptr)` + free_align: *const fn (?*anyopaque) callconv(.c) void, + + // ---- Thread, :25-27 + /// :25 `void* (*)(const char *tname, uint32_t tprio, uint32_t tstack_size, void (*start_routine)(void const *), void *sr_arg)` + thread_create: *const fn ([*:0]const u8, u32, u32, StartRoutine, ?*anyopaque) callconv(.c) ?*anyopaque, + /// :26 `int (*)(void *thread_handle)` + thread_cancel: *const fn (?*anyopaque) callconv(.c) c_int, + /// :27 `void (*)(void)` + thread_yield: *const fn () callconv(.c) void, + + // ---- Sleeps, :30-32 + /// :30 `unsigned int (*)(unsigned int mseconds)` + msleep: *const fn (c_uint) callconv(.c) c_uint, + /// :31 `unsigned int (*)(unsigned int useconds)` + usleep: *const fn (c_uint) callconv(.c) c_uint, + /// :32 `unsigned int (*)(unsigned int seconds)` + sleep: *const fn (c_uint) callconv(.c) c_uint, + + // ---- Blocking non-sleepable delay, :35 + /// :35 `unsigned int (*)(unsigned int number)` + blocking_delay: *const fn (c_uint) callconv(.c) c_uint, + + // ---- Queue, :38-43 + /// :38 `int (*)(void * queue_handle, void *item, int timeout)` + queue_item: *const fn (?*anyopaque, ?*const anyopaque, c_int) callconv(.c) c_int, + /// :39 `void* (*)(uint32_t qnum_elem, uint32_t qitem_size)` + create_queue: *const fn (u32, u32) callconv(.c) ?*anyopaque, + /// :40 `int (*)(void * queue_handle, void *item, int timeout)` + dequeue_item: *const fn (?*anyopaque, ?*anyopaque, c_int) callconv(.c) c_int, + /// :41 `int (*)(void * queue_handle)` + queue_msg_waiting: *const fn (?*anyopaque) callconv(.c) c_int, + /// :42 `int (*)(void * queue_handle)` + destroy_queue: *const fn (?*anyopaque) callconv(.c) c_int, + /// :43 `int (*)(void * queue_handle)` + reset_queue: *const fn (?*anyopaque) callconv(.c) c_int, + + // ---- Mutex, :46-49. Note that unlock comes *first*. + /// :46 `int (*)(void * mutex_handle)` + unlock_mutex: *const fn (?*anyopaque) callconv(.c) c_int, + /// :47 `void* (*)(void)` + create_mutex: *const fn () callconv(.c) ?*anyopaque, + /// :48 `int (*)(void * mutex_handle, int timeout_ms)` + lock_mutex: *const fn (?*anyopaque, c_int) callconv(.c) c_int, + /// :49 `int (*)(void * mutex_handle)` + destroy_mutex: *const fn (?*anyopaque) callconv(.c) c_int, + + // ---- Semaphore, :52-56. `post` precedes `create`, as with the mutex. + /// :52 `int (*)(void * semaphore_handle)` + post_semaphore: *const fn (?*anyopaque) callconv(.c) c_int, + /// :53 `int (*)(void * semaphore_handle)` + post_semaphore_from_isr: *const fn (?*anyopaque) callconv(.c) c_int, + /// :54 `void* (*)(int maxCount)` + create_semaphore: *const fn (c_int) callconv(.c) ?*anyopaque, + /// :55 `int (*)(void * semaphore_handle, int timeout_ms)` + get_semaphore: *const fn (?*anyopaque, c_int) callconv(.c) c_int, + /// :56 `int (*)(void * semaphore_handle)` + destroy_semaphore: *const fn (?*anyopaque) callconv(.c) c_int, + + // ---- Timer, :59-61. `stop` precedes `start`. + /// :59 `int (*)(void *timer_handle)` + timer_stop: *const fn (?*anyopaque) callconv(.c) c_int, + /// :60 `void* (*)(const char *name, int duration_ms, int type, void (*timeout_handler)(void *), void *arg)` + timer_start: *const fn ([*:0]const u8, c_int, c_int, TimerHandler, ?*anyopaque) callconv(.c) ?*anyopaque, + /// :61 `uint64_t (*)(void)` + get_time_ms: *const fn () callconv(.c) u64, + + // ---- Mempool, :65-68, present because H_USE_MEMPOOL is defined. See the file header. + /// :65 `void* (*)(void)` + create_lock_mempool: *const fn () callconv(.c) ?*anyopaque, + /// :66 `void (*)(void *lock_handle)` + lock_mempool: *const fn (?*anyopaque) callconv(.c) void, + /// :67 `void (*)(void *lock_handle)` + unlock_mempool: *const fn (?*anyopaque) callconv(.c) void, + /// :68 `void (*)(void *lock_handle)` + destroy_lock_mempool: *const fn (?*anyopaque) callconv(.c) void, + + // ---- GPIO, :72-79 + /// :72 `int (*)(void* gpio_port, uint32_t gpio_num, uint32_t mode)` + config_gpio: *const fn (?*anyopaque, u32, u32) callconv(.c) c_int, + /// :73 `int (*)(void* gpio_port, uint32_t gpio_num, uint32_t intr_type, void (*gpio_isr_handler)(void* arg), void *arg)` + config_gpio_as_interrupt: *const fn (?*anyopaque, u32, u32, IsrHandler, ?*anyopaque) callconv(.c) c_int, + /// :74 `int (*)(void* gpio_port, uint32_t gpio_num)` + teardown_gpio_interrupt: *const fn (?*anyopaque, u32) callconv(.c) c_int, + /// :75 `int (*)(void* gpio_port, uint32_t gpio_num)` + read_gpio: *const fn (?*anyopaque, u32) callconv(.c) c_int, + /// :76 `int (*)(void* gpio_port, uint32_t gpio_num, uint32_t value)` + write_gpio: *const fn (?*anyopaque, u32, u32) callconv(.c) c_int, + /// :77 `int (*)(void* gpio_port, uint32_t gpio_num, uint32_t pull_value, uint32_t enable)` + pull_gpio: *const fn (?*anyopaque, u32, u32, u32) callconv(.c) c_int, + /// :78 `int (*)(void* gpio_port, uint32_t gpio_num, uint32_t hold_value)` + hold_gpio: *const fn (?*anyopaque, u32, u32) callconv(.c) c_int, + /// :79 `int (*)(void)` + get_host_wakeup_or_reboot_reason: *const fn () callconv(.c) c_int, + + // ---- All transports, :81-82 + /// :81 `void * (*)(void)` + bus_init: *const fn () callconv(.c) ?*anyopaque, + /// :82 `int (*)(void*)` + bus_deinit: *const fn (?*anyopaque) callconv(.c) c_int, + + // ---- :84-88 + /// :84 `int (*)(void *transfer_context)` - SPI only; ESP-IDF leaves this null under SDIO. + do_bus_transfer: *const fn (?*anyopaque) callconv(.c) c_int, + /// :85 `int (*)(int32_t event_id, void* event_data, size_t event_data_size, uint32_t ticks_to_wait)` + event_wifi_post: *const fn (i32, ?*anyopaque, usize, u32) callconv(.c) c_int, + /// :87 `void (*)(int level, const char *tag, const char *format, ...)` + printf: *const fn (c_int, [*:0]const u8, [*:0]const u8, ...) callconv(.c) void, + /// :88 `void (*)(void)` + hosted_init_hook: *const fn () callconv(.c) void, + + // ---- Transport - SDIO, :91-97 + /// :91 `int (*)(void *ctx, bool show_config)` + sdio_card_init: *const fn (?*anyopaque, bool) callconv(.c) c_int, + /// :92 `int (*)(void*ctx)` + sdio_card_deinit: *const fn (?*anyopaque) callconv(.c) c_int, + /// :93 `int (*)(void *ctx, uint32_t reg, uint8_t *data, uint16_t size, bool lock_required)` + sdio_read_reg: *const fn (?*anyopaque, u32, [*]u8, u16, bool) callconv(.c) c_int, + /// :94 same + sdio_write_reg: *const fn (?*anyopaque, u32, [*]u8, u16, bool) callconv(.c) c_int, + /// :95 same + sdio_read_block: *const fn (?*anyopaque, u32, [*]u8, u16, bool) callconv(.c) c_int, + /// :96 same + sdio_write_block: *const fn (?*anyopaque, u32, [*]u8, u16, bool) callconv(.c) c_int, + /// :97 `int (*)(void *ctx, uint32_t ticks_to_wait)` + sdio_wait_slave_intr: *const fn (?*anyopaque, u32) callconv(.c) c_int, + + // ---- Transport - SPI HD, :100-105 + /// :100 `int (*)(uint32_t reg, uint32_t *data, int poll, bool lock_required)` + spi_hd_read_reg: *const fn (u32, *u32, c_int, bool) callconv(.c) c_int, + /// :101 `int (*)(uint32_t reg, uint32_t *data, bool lock_required)` + spi_hd_write_reg: *const fn (u32, *u32, bool) callconv(.c) c_int, + /// :102 `int (*)(uint8_t *data, uint16_t size, bool lock_required)` + spi_hd_read_dma: *const fn ([*]u8, u16, bool) callconv(.c) c_int, + /// :103 same + spi_hd_write_dma: *const fn ([*]u8, u16, bool) callconv(.c) c_int, + /// :104 `int (*)(uint32_t data_lines)` + spi_hd_set_data_lines: *const fn (u32) callconv(.c) c_int, + /// :105 `int (*)(void)` + spi_hd_send_cmd9: *const fn () callconv(.c) c_int, + + // ---- Transport - UART, :108-110 + /// :108 `int (*)(void *ctx, uint8_t *data, uint16_t size)` + uart_read: *const fn (?*anyopaque, [*]u8, u16) callconv(.c) c_int, + /// :109 same + uart_write: *const fn (?*anyopaque, [*]u8, u16) callconv(.c) c_int, + /// :110 `int (*)(void *ctx)` + uart_flush_input: *const fn (?*anyopaque) callconv(.c) c_int, + + /// :112 `int (*)(void)` + restart_host: *const fn () callconv(.c) c_int, + + /// :114 `int (*)(uint32_t power_save_type, void* gpio_port, uint32_t gpio_num, int level)` + config_host_power_save_hal_impl: *const fn (u32, ?*anyopaque, u32, c_int) callconv(.c) c_int, + /// :115 `int (*)(uint32_t power_save_type)` + start_host_power_save_hal_impl: *const fn (u32) callconv(.c) c_int, + /// :116 `int (*)(esp_event_base_t event_base, int32_t event_id, void* event_data, size_t event_data_size, uint32_t ticks_to_wait)` + event_post: *const fn (EventBase, i32, ?*anyopaque, usize, u32) callconv(.c) c_int, +}; + +/// `struct hosted_config_t`, `esp_hosted_os_abstraction.h:119-121`. +pub const HostedConfig = extern struct { + funcs: *const HostedOsiFuncs, +}; + +/// The three numbers the C side must agree on. Measured from C with the force-include in place; +/// the build re-measures and compares, so this is a contract and not a comment. +pub const layout_check = struct { + pub const sizeof: usize = 284; + pub const offset_config_gpio: usize = 148; + pub const offset_event_post: usize = 280; +}; + +comptime { + if (@sizeOf(usize) != 4) @compileError( + "this layout is rv32-specific: 71 pointers at 4 bytes each. Re-measure offsetof on any other target.", + ); + assert(@sizeOf(HostedOsiFuncs) == layout_check.sizeof); + assert(@offsetOf(HostedOsiFuncs, "config_gpio") == layout_check.offset_config_gpio); + assert(@offsetOf(HostedOsiFuncs, "event_post") == layout_check.offset_event_post); + // Every member is one pointer, so the count is derivable and worth asserting: a field + // accidentally deleted or duplicated changes this even when the size happens to survive. + assert(std.meta.fields(HostedOsiFuncs).len == 71); + assert(@sizeOf(HostedOsiFuncs) == 71 * @sizeOf(usize)); +} + +// ============================================================================ the exported table + +/// The table itself. `HOSTED_CONFIG_INIT_DEFAULT` points `g_h.funcs` here +/// (`esp_hosted_os_abstraction.h:125-127`), and `port_esp_hosted_host_os.c:938` is the definition +/// this replaces. +pub export const g_hosted_osi_funcs: HostedOsiFuncs = .{ + .memcpy = hostedMemcpy, + .memset = hostedMemset, + .malloc = hostedMalloc, + .calloc = hostedCalloc, + .free = hostedFree, + .realloc = hostedRealloc, + .malloc_align = hostedMallocAlign, + .free_align = hostedFreeAlign, + + .thread_create = hostedThreadCreate, + .thread_cancel = hostedThreadCancel, + .thread_yield = hostedThreadYield, + + .msleep = hostedMsleep, + .usleep = hostedUsleep, + .sleep = hostedSleep, + .blocking_delay = hostedBlockingDelay, + + .queue_item = hostedQueueItem, + .create_queue = hostedCreateQueue, + .dequeue_item = hostedDequeueItem, + .queue_msg_waiting = hostedQueueMsgWaiting, + .destroy_queue = hostedDestroyQueue, + .reset_queue = hostedResetQueue, + + .unlock_mutex = hostedUnlockMutex, + .create_mutex = hostedCreateMutex, + .lock_mutex = hostedLockMutex, + .destroy_mutex = hostedDestroyMutex, + + .post_semaphore = hostedPostSemaphore, + .post_semaphore_from_isr = hostedPostSemaphoreFromIsr, + .create_semaphore = hostedCreateSemaphore, + .get_semaphore = hostedGetSemaphore, + .destroy_semaphore = hostedDestroySemaphore, + + .timer_stop = hostedTimerStop, + .timer_start = hostedTimerStart, + .get_time_ms = hostedGetTimeMs, + + .create_lock_mempool = hostedCreateLockMempool, + .lock_mempool = hostedLockMempool, + .unlock_mempool = hostedUnlockMempool, + .destroy_lock_mempool = hostedDestroyLockMempool, + + .config_gpio = hostedConfigGpio, + .config_gpio_as_interrupt = hostedConfigGpioAsInterrupt, + .teardown_gpio_interrupt = hostedTeardownGpioInterrupt, + .read_gpio = hostedReadGpio, + .write_gpio = hostedWriteGpio, + .pull_gpio = hostedPullGpio, + .hold_gpio = hostedHoldGpio, + .get_host_wakeup_or_reboot_reason = hostedGetWakeupReason, + + .bus_init = hostedBusInit, + .bus_deinit = hostedBusDeinit, + + .do_bus_transfer = stubDoBusTransfer, + .event_wifi_post = hostedEventWifiPost, + .printf = stubPrintf, + .hosted_init_hook = hostedInitHook, + + .sdio_card_init = hostedSdioCardInit, + .sdio_card_deinit = hostedSdioCardDeinit, + .sdio_read_reg = hostedSdioReadReg, + .sdio_write_reg = hostedSdioWriteReg, + .sdio_read_block = hostedSdioReadBlock, + .sdio_write_block = hostedSdioWriteBlock, + .sdio_wait_slave_intr = hostedSdioWaitSlaveIntr, + + .spi_hd_read_reg = stubSpiHdReadReg, + .spi_hd_write_reg = stubSpiHdWriteReg, + .spi_hd_read_dma = stubSpiHdReadDma, + .spi_hd_write_dma = stubSpiHdWriteDma, + .spi_hd_set_data_lines = stubSpiHdSetDataLines, + .spi_hd_send_cmd9 = stubSpiHdSendCmd9, + + .uart_read = stubUartRead, + .uart_write = stubUartWrite, + .uart_flush_input = stubUartFlushInput, + + .restart_host = hostedRestartHost, + + .config_host_power_save_hal_impl = stubConfigHostPowerSave, + .start_host_power_save_hal_impl = stubStartHostPowerSave, + .event_post = hostedEventPost, +}; + +/// `extern struct hosted_config_t g_h;` (`esp_hosted_os_abstraction.h:129`). Statically +/// initialised, because C reads `g_h.funcs->...` and nothing guarantees `install` ran first - it is +/// the *state* behind the functions that needs installing, not the pointer to them. +pub export var g_h: HostedConfig = .{ .funcs = &g_hosted_osi_funcs }; + +// ============================================================================ installed state + +/// Board wiring and sizing. Compile-time so the static footprint is a build-time number. +pub const Config = struct { + /// The C6's reset/enable pin. GPIO54 on this board (`sdkconfig:4557`, + /// `CONFIG_ESP_HOSTED_GPIO_SLAVE_RESET_SLAVE=54`). + /// + /// It has an external pull-up, so the *released* state is the one the pull-up wins. ESP-Hosted + /// drives `H_RESET_VAL_ACTIVE` last (`sdio_drv.c:1651-1657`), and with + /// `CONFIG_ESP_HOSTED_RESET_GPIO_ACTIVE_LOW` unset - which is how the working IDF build on this + /// board was configured - `H_RESET_VAL_ACTIVE` is `H_GPIO_HIGH` + /// (`port_esp_hosted_host_config.h:445-451`). So the sequence is high, low, high: a reset pulse + /// that ends released. Invert that and the radio stays in reset for ever. + reset_pin: u8 = 54, + + /// CLIC external line for the GPIO interrupt aggregate (`hal.intr.Source.gpio_intr0`). + gpio_clic_line: u5 = 20, + /// CLIC external line for the SDMMC host, which is where the C6's D1 slave interrupt arrives. + sdio_clic_line: u5 = 21, + + /// Software timer slots. ESP-Hosted arms at most three at once: the slave-unresponsive timer + /// (`transport_drv.c:188`), a per-request asynchronous RPC timeout (`rpc_core.c:215`), and the + /// power-save timer. Four leaves one spare and costs 96 bytes. + timer_slots: usize = 4, + + /// Pads whose interrupt can be registered at once. The SDIO transport registers none; SPI + /// registers two. Four is generous and costs 48 bytes. + gpio_isr_slots: usize = 4, + + /// Wait for the C6's D1 slave interrupt through the CLIC, or poll for it. + /// + /// `true`. The reason it was `false` is worth keeping written down, because it was a + /// misdiagnosis rather than a hardware limit. + /// + /// Every attempt printed `MARK PORT_SDIO_LAPSE ... intmask=0x00000000`, and that was read as + /// "the unmask does not stick". It never said that: the print happens *after* + /// `disarmSdioLine()`, which had just written that zero on purpose, and the other witness - + /// `hal.sdmmc.interruptDiagnostics` - runs on the application task, which is never inside an + /// arming window. `hal.sdmmc.armSlaveInterrupt` now reads INTMASK back inside the same masked + /// region as the store, so the claim is finally testable: `stuck=` on `MARK PORT_SDIO_ARM`. + /// + /// What was really missing is `takeInterruptControl`. Nothing in this build had ever called + /// `hal.intr.init()` - `examples/intrcheck.zig` and `examples/portcheck.zig` do, + /// `examples/http.zig` and `examples/radio.zig` do not, and nothing under `src/` did either - + /// so mtvec still belonged to the bootloader, the threshold was never opened, and mstatus.MIE + /// was never this image's decision. A CLIC line enabled in that state either cannot be + /// delivered at all, which is a LAPSE every window for ever, or is delivered *outside this + /// image*, which is the "board goes silent right after Open data path at slave" that was + /// blamed on a storm. + /// + /// Not verified on hardware by the author of this change. Two nets remain under it: the + /// bounded re-look (`sdio_relook_ms`) carries the transport through any window the interrupt + /// misses, and `sdio_foreign_limit` consecutive unexplained handler entries abandon the line + /// for `sdioPoll` permanently. Set this to `false` to isolate a regression against the proven + /// polling path; nothing else has to change, and with it false the CLIC is not touched at all. + sdio_use_interrupt: bool = true, +}; + +pub const config: Config = .{}; + +const State = struct { + io: Io = undefined, + /// The allocator handed to `install`. Used directly for OS-object handles, and wrapped by + /// `cheap` for everything C allocates. + gpa: Allocator = undefined, + cheap: hheap.CHeap = undefined, + timers: os.TimerService(config.timer_slots) = .{}, + installed: bool = false, + + /// The bus context `_h_bus_init` hands to C and C hands back to every `_h_sdio_*` call. Its + /// *identity* is all that matters - ESP-IDF returns `&context`, a file-static - so this is a + /// single static object and a null `ctx` from C is a real error rather than a second bus. + bus: BusContext = .{}, + + /// Called with every event ESP-Hosted posts. Association and DHCP-relevant events arrive here. + on_event: ?*const fn (Event) void = null, + + /// Deferred wake word for the SDIO slave interrupt. The ISR bumps it and wakes; the waiter + /// futex-waits on it. + sdio_intr_epoch: std.atomic.Value(u32) = .init(0), + + /// RINTSTS and IDSTS as `sdioDispatch` saw them at entry. Both are sticky, so reading them + /// after the handler has disarmed loses nothing. INTMASK is *not* sticky and is deliberately + /// absent here: the disarm has just rewritten it, so a handler-entry read of it could only ever + /// return the disarmed value. What the mask really was is `sdio_armed_intmask`. + sdio_intr_rintsts: std.atomic.Value(u32) = .init(0), + sdio_intr_idsts: std.atomic.Value(u32) = .init(0), + + /// INTMASK and MINTSTS as `hal.sdmmc.armSlaveInterrupt` read them back, inside the same masked + /// region as the store that armed them. Written and read only by the waiting task, so plain + /// words rather than atomics. + sdio_armed_intmask: u32 = 0, + sdio_armed_mintsts: u32 = 0, + + /// Consecutive handler entries whose cause was not the card interrupt. Reset by any real one. + /// At `sdio_foreign_limit` the wait stops using the interrupt at all. + sdio_intr_foreign: u32 = 0, + + /// Arming windows that lapsed with no handler entry. **At idle this is the normal state and + /// says nothing is wrong**: the C6 has nothing to report, so no interrupt arrives inside + /// `sdio_relook_ms`, the re-look finds nothing either, and the wait goes round again. It is + /// counted and printed because a *rising* count with frames flowing is how the re-look + /// carrying the transport announces itself. + sdio_intr_lapses: u32 = 0, + + /// Consecutive lapsed windows in which the re-look then found the card *already calling* - + /// the pad low or the latch set. That is the failure that matters, and it is the only reading + /// that separates "the interrupt is not being delivered" from "the card is quiet": an idle + /// card lapses for free, a calling card whose interrupt did not arrive costs a real frame up + /// to `sdio_relook_ms` of latency. + /// + /// Reset by any wake the handler really delivered. At `sdio_missed_limit` the wait gives the + /// line up for `sdioPoll` permanently, which is what keeps the interrupt path from being + /// strictly worse than the 1 ms poll it replaces. + sdio_intr_missed: u32 = 0, + + /// MINTSTS and RINTSTS as `sdioDispatch` read them, *before* it disarmed. MINTSTS is + /// `RINTSTS & INTMASK` and the disarm zeroes it, so this is the only place its value at the + /// moment of delivery survives - and it is the direct answer to "does MINTSTS ever show this + /// slot's bit". + sdio_intr_mintsts: std.atomic.Value(u32) = .init(0), + + /// Remaining diagnostic lines, one budget per failure mode. See `sdioMark`. + sdio_foreign_marks: u32 = 0, + sdio_lapse_marks: u32 = 0, + sdio_missed_marks: u32 = 0, + sdio_arm_marks: u32 = 0, + sdio_wake_marks: u32 = 0, + + /// Arming windows completed, for the periodic tally. Every budgeted MARK above eventually goes + /// quiet; this one does not, because "is the interrupt or the re-look carrying the transport" + /// is a question that stays interesting for the whole run. + sdio_windows: u32 = 0, + + /// GPIO ISR registrations, indexed arbitrarily. + gpio_isrs: [config.gpio_isr_slots]GpioIsr = @splat(.{}), + + /// `_h_sleep` calls. In the file set build.zig compiles this counts exactly one thing: the + /// two `if (!is_rpc_lib_ready()) _h_sleep(1)` loops at the head of `rpc_rx_thread` and + /// `rpc_tx_thread` (rpc_core.c:482-485, :543-547). The tree's only other `_h_sleep` callers + /// are transport_drv.c:693, which is followed by `assert(0!=0)`, and stats.c:115 in + /// `raw_tp_tx_task`, which is never created with TEST_RAW_TP off. + /// + /// So a count that keeps *growing* while a synchronous RPC request is outstanding means the + /// RPC lib state is not READY and the request will never be transmitted - the failure that + /// otherwise looks exactly like a coprocessor that does not answer. Two per second while + /// stuck, and it costs one add. + hosted_sleep_calls: u32 = 0, + + /// Loud-stub call count. Nonzero after a run means a path nobody implemented was taken. + stub_calls: u32 = 0, +}; + +const GpioIsr = struct { + pin: u8 = 0xFF, + handler: ?IsrHandler = null, + arg: ?*anyopaque = null, +}; + +const BusContext = struct { + /// `hosted_sdio_init` creates this and every `SDIO_LOCK` takes it + /// (`port_esp_hosted_host_sdio.c:36-42, 395`). + lock: os.Mutex = .{}, + up: bool = false, +}; + +var state: State = .{}; + +/// An event ESP-Hosted posted. `base` distinguishes `WIFI_EVENT` (via `_h_event_wifi_post`) from +/// `ESP_HOSTED_EVENT` and anything else (via `_h_event_post`). +pub const Event = struct { + pub const Base = union(enum) { + wifi, + /// The `esp_event_base_t` string C passed, which is a pointer to a string literal owned by + /// the C side and valid for the lifetime of the program. + named: EventBase, + }; + base: Base, + id: i32, + /// Borrowed for the duration of the callback only. ESP-IDF's `esp_event_post` copies; + /// this does not, so a handler that needs the data past its return must copy it. + data: ?[]const u8, +}; + +/// Bring the table's state up. Idempotent. +/// +/// After this returns, C may call anything in `g_h.funcs`. Note what it does *not* do: it does not +/// start a scheduler and it does not touch the radio. ESP-Hosted's own `esp_hosted_init` does that, +/// and the tasks it spawns through `_h_thread_create` first execute when the calling context next +/// blocks - `io.async` assigns a slot and marks it ready, it does not preempt. A caller that +/// installs, initialises ESP-Hosted and then never blocks will see nothing happen. +pub fn install(io: Io, gpa: Allocator) void { + state.io = io; + state.gpa = gpa; + state.cheap = .{ .gpa = gpa }; + state.installed = true; + // The timer service owns one task; start it eagerly so `_h_timer_start` cannot fail for want + // of a scheduler. + if (!state.timers.start(io, gpa)) note("MARK PORT_TIMER_SERVICE_FAIL\r\n", .{}); +} + +/// Register the application's event sink. Association, disconnection and the slave's own lifecycle +/// events arrive here; this is not a reimplementation of `esp_event`, it is one callback. +pub fn setEventHandler(handler: ?*const fn (Event) void) void { + state.on_event = handler; +} + +/// Diagnostics for a hardware self-test: heap use, whether any loud stub was reached, and whether +/// ESP-Hosted's RPC threads are stuck in their not-ready loop. See `State.hosted_sleep_calls`. +pub fn stats() struct { + bytes_live: usize, + bytes_reserved: usize, + peak_reserved: usize, + blocks_live: usize, + alloc_failures: usize, + stub_calls: u32, + hosted_sleep_calls: u32, +} { + return .{ + .bytes_live = state.cheap.bytes_live, + .bytes_reserved = state.cheap.bytes_reserved, + .peak_reserved = state.cheap.peak_reserved, + .blocks_live = state.cheap.blocks_live, + .alloc_failures = state.cheap.failures, + .stub_calls = state.stub_calls, + .hosted_sleep_calls = state.hosted_sleep_calls, + }; +} + +/// The SDIO card-interrupt path's counters, for a heartbeat that wants to say whether the radio is +/// being woken or polled. Every field is a running total, none is reset by anything here. +/// +/// `epoch` is handler entries. `foreign` is *consecutive* entries whose cause was not the card +/// interrupt - at `sdio_foreign_limit` the wait abandons the interrupt for `sdioPoll`, so a +/// non-zero `foreign` with a growing `epoch` means the line is being taken for the wrong reason. +/// `lapses` is arming windows that produced no entry at all; at idle that is the resting state and +/// costs nothing. `missed` is the subset of those whose re-look then found the card already +/// calling, which is the one that matters - at `sdio_missed_limit` the wait abandons the interrupt +/// too. `rintsts`/`idsts` are what the last handler entry saw; `armed_intmask` is what INTMASK read +/// back at the last arm, which is the only reading of that register that means anything. +pub fn sdioStats() struct { + epoch: u32, + foreign: u32, + lapses: u32, + missed: u32, + rintsts: u32, + idsts: u32, + armed_intmask: u32, +} { + return .{ + .epoch = state.sdio_intr_epoch.load(.acquire), + .foreign = state.sdio_intr_foreign, + .lapses = state.sdio_intr_lapses, + .missed = state.sdio_intr_missed, + .rintsts = state.sdio_intr_rintsts.load(.acquire), + .idsts = state.sdio_intr_idsts.load(.acquire), + .armed_intmask = state.sdio_armed_intmask, + }; +} + +inline fn currentIo() Io { + assert(state.installed); + return state.io; +} + +// ============================================================================ 1. memory + +fn hostedMemcpy(dest: ?*anyopaque, src: ?*const anyopaque, size: u32) callconv(.c) ?*anyopaque { + // ESP-IDF asserts on a null pointer with a nonzero size (port_esp_hosted_host_os.c:67-76); the + // same condition, as a Zig assertion. + if (size == 0) return dest; + const d: [*]u8 = @ptrCast(dest.?); + const s: [*]const u8 = @ptrCast(src.?); + @memcpy(d[0..size], s[0..size]); + return dest; +} + +fn hostedMemset(buf: ?*anyopaque, val: c_int, len: usize) callconv(.c) ?*anyopaque { + if (len == 0) return buf; + const b: [*]u8 = @ptrCast(buf.?); + @memset(b[0..len], @truncate(@as(c_uint, @bitCast(val)))); + return buf; +} + +fn hostedMalloc(size: usize) callconv(.c) ?*anyopaque { + assert(state.installed); + return @ptrCast(state.cheap.malloc(size)); +} + +fn hostedCalloc(blk_no: usize, size: usize) callconv(.c) ?*anyopaque { + assert(state.installed); + return @ptrCast(state.cheap.calloc(blk_no, size)); +} + +fn hostedFree(ptr: ?*anyopaque) callconv(.c) void { + assert(state.installed); + state.cheap.free(@ptrCast(ptr)); +} + +fn hostedRealloc(mem: ?*anyopaque, newsize: usize) callconv(.c) ?*anyopaque { + assert(state.installed); + return @ptrCast(state.cheap.realloc(@ptrCast(mem), newsize)); +} + +/// `_h_malloc_align(size, align)`. ESP-IDF routes this to `heap_caps_aligned_alloc` with +/// DMA-capable caps (`port_esp_hosted_host_os.c:128-143`) because IDF's SDMMC driver DMAs straight +/// out of the caller's buffer. +/// +/// Ours does not: `hal.sdmmc` bounces every CMD53 through its own 64-byte-aligned buffer reached +/// through the non-cacheable alias, and memcpy's to and from the caller's slice. So the alignment +/// is honoured - it costs 64 bytes a buffer and callers may reasonably rely on it - but nothing +/// downstream needs it, and `_h_malloc` would do. +fn hostedMallocAlign(size: usize, alignment: usize) callconv(.c) ?*anyopaque { + assert(state.installed); + // ESP-Hosted only ever asks for 4, 32 or 64 (HOSTED_MEM_ALIGNMENT_*, + // port_esp_hosted_host_os.h:93-95). A non-power-of-two would silently corrupt the header + // arithmetic, so refuse it. + if (alignment == 0 or !std.math.isPowerOfTwo(alignment) or alignment > hheap.CHeap.max_alignment) { + note("MARK PORT_BAD_ALIGN %u\r\n", .{@as(u32, @intCast(alignment))}); + return null; + } + return @ptrCast(state.cheap.mallocAligned(size, alignment)); +} + +/// One header format for both `_h_free` and `_h_free_align`, because ESP-IDF has one too: its +/// `hosted_free_align` is a plain `free` (`port_esp_hosted_host_os.c:145-148`), and mixing the two +/// is legal in the tree - `sdio_drv.c:353` frees with `_h_free_align` a buffer that +/// `transport_util.c:14` allocated with `_h_malloc_align`, while `HOSTED_FREE` uses `_h_free` +/// throughout. +fn hostedFreeAlign(ptr: ?*anyopaque) callconv(.c) void { + assert(state.installed); + state.cheap.free(@ptrCast(ptr)); +} + +// ============================================================================ 2. sync + +fn hostedCreateMutex() callconv(.c) ?*anyopaque { + assert(state.installed); + const m = state.gpa.create(os.Mutex) catch return null; + m.* = .{}; + return @ptrCast(m); +} + +fn hostedLockMutex(handle: ?*anyopaque, timeout_ms: c_int) callconv(.c) c_int { + const m: *os.Mutex = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return m.lock(currentIo(), .fromMillis(timeout_ms)); +} + +fn hostedUnlockMutex(handle: ?*anyopaque) callconv(.c) c_int { + const m: *os.Mutex = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return m.unlock(currentIo()); +} + +fn hostedDestroyMutex(handle: ?*anyopaque) callconv(.c) c_int { + const m: *os.Mutex = @ptrCast(@alignCast(handle orelse return ret.invalid)); + state.gpa.destroy(m); + return ret.ok; +} + +fn hostedCreateSemaphore(max_count: c_int) callconv(.c) ?*anyopaque { + assert(state.installed); + const s = state.gpa.create(os.Semaphore) catch return null; + s.* = .init(if (max_count > 0) @intCast(max_count) else 1); + return @ptrCast(s); +} + +fn hostedPostSemaphore(handle: ?*anyopaque) callconv(.c) c_int { + const s: *os.Semaphore = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return s.post(currentIo()); +} + +/// See `hosted_os.Semaphore.postFromIsr` for what "from ISR" can and cannot mean here. +fn hostedPostSemaphoreFromIsr(handle: ?*anyopaque) callconv(.c) c_int { + const s: *os.Semaphore = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return s.postFromIsr(state.io); +} + +fn hostedGetSemaphore(handle: ?*anyopaque, timeout_ms: c_int) callconv(.c) c_int { + const s: *os.Semaphore = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return s.wait(currentIo(), .fromMillis(timeout_ms)); +} + +fn hostedDestroySemaphore(handle: ?*anyopaque) callconv(.c) c_int { + const s: *os.Semaphore = @ptrCast(@alignCast(handle orelse return ret.invalid)); + state.gpa.destroy(s); + return ret.ok; +} + +fn hostedCreateQueue(qnum_elem: u32, qitem_size: u32) callconv(.c) ?*anyopaque { + assert(state.installed); + if (qnum_elem == 0 or qitem_size == 0) return null; + return @ptrCast(os.Queue.create(state.gpa, qnum_elem, qitem_size)); +} + +fn hostedQueueItem(handle: ?*anyopaque, item: ?*const anyopaque, timeout: c_int) callconv(.c) c_int { + const q: *os.Queue = @ptrCast(@alignCast(handle orelse return ret.invalid)); + const p: [*]const u8 = @ptrCast(item orelse return ret.invalid); + // `_h_queue_item`'s timeout reaches xQueueSendToBack unconverted, so its units are ticks; every + // caller passes HOSTED_BLOCK_MAX or 0, both of which mean the same thing in either dialect. + return q.send(currentIo(), p, .fromMillis(timeout)); +} + +fn hostedDequeueItem(handle: ?*anyopaque, item: ?*anyopaque, timeout: c_int) callconv(.c) c_int { + const q: *os.Queue = @ptrCast(@alignCast(handle orelse return ret.invalid)); + const p: [*]u8 = @ptrCast(item orelse return ret.invalid); + // Seconds, not milliseconds, on the positive branch. See `hosted_os.Wait.fromQueueTimeout`. + return q.receive(currentIo(), p, .fromQueueTimeout(timeout)); +} + +fn hostedQueueMsgWaiting(handle: ?*anyopaque) callconv(.c) c_int { + const q: *os.Queue = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return q.waiting(currentIo()); +} + +fn hostedDestroyQueue(handle: ?*anyopaque) callconv(.c) c_int { + const q: *os.Queue = @ptrCast(@alignCast(handle orelse return ret.invalid)); + q.destroy(currentIo(), state.gpa); + return ret.ok; +} + +fn hostedResetQueue(handle: ?*anyopaque) callconv(.c) c_int { + const q: *os.Queue = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return q.reset(currentIo()); +} + +/// The mempool lock. `H_USE_MEMPOOL` is 1 in this board's configuration, so these four must not be +/// null even though the version of `common/mempool/mempool.c` in this tree does not call them. +/// +/// ESP-IDF uses a `portMUX_TYPE` spinlock and `portENTER_CRITICAL` +/// (`port_esp_hosted_host_os.c:602-643`), which on a multi-core preemptive kernel means "take the +/// spinlock and disable interrupts". On one core with a cooperative scheduler the spinlock half is +/// vacuous - there is no other core to contend with - and the interrupt half is the whole content. +/// So the handle is `hal.intr`'s nesting mask guard, and the critical section is exactly as long as +/// interrupts are off. +const MempoolLock = struct { + guard: hal.clkrst.Guard = undefined, + held: bool = false, +}; + +fn hostedCreateLockMempool() callconv(.c) ?*anyopaque { + assert(state.installed); + const l = state.gpa.create(MempoolLock) catch return null; + l.* = .{}; + return @ptrCast(l); +} + +fn hostedLockMempool(handle: ?*anyopaque) callconv(.c) void { + const l: *MempoolLock = @ptrCast(@alignCast(handle orelse return)); + l.guard = hal.intr.mask(); + l.held = true; +} + +fn hostedUnlockMempool(handle: ?*anyopaque) callconv(.c) void { + const l: *MempoolLock = @ptrCast(@alignCast(handle orelse return)); + if (!l.held) return; + l.held = false; + l.guard.release(); +} + +fn hostedDestroyLockMempool(handle: ?*anyopaque) callconv(.c) void { + const l: *MempoolLock = @ptrCast(@alignCast(handle orelse return)); + state.gpa.destroy(l); +} + +// ============================================================================ 3. threads + +/// ESP-Hosted spawns **seven** tasks on the SDIO transport, and their requested stacks are the +/// single largest memory claim in the whole port: +/// +/// sdio_rx_buf RX_BUF_TASK_STACK_SIZE sdio_drv.c:1542 (= CONFIG_ESP_HOSTED_DFLT_TASK_STACK) +/// sdio_read DFLT_TASK_STACK_SIZE sdio_drv.c:1545 +/// sdio_process_rx DFLT_TASK_STACK_SIZE sdio_drv.c:1548 +/// sdio_write DFLT_TASK_STACK_SIZE sdio_drv.c:1551 +/// rpc_rx RPC_TASK_STACK_SIZE rpc_core.c:578 +/// rpc_tx RPC_TASK_STACK_SIZE rpc_core.c:580 +/// rpc_supp_cb RPC_TASK_STACK_SIZE rpc_wrap.c:2398 +/// +/// `DFLT_TASK_STACK_SIZE` and `RPC_TASK_STACK_SIZE` are both `5*1024` +/// (`port_esp_hosted_host_os.h:64-67`), and ESP-IDF's `xTaskCreate` takes bytes, so the ask is +/// 35 KB. Plus this port's timer service task, plus the main context, that is nine slots. +/// +/// The requested size is **ignored**, and that is not laziness: `std.Io.async` has no stack-size +/// parameter, and the runtime takes the first free slot from a pool whose slots are all declared at +/// one size. The number to declare is therefore the worst case over all seven, which is what the +/// caller of `install` decides when it builds its `Runtime`. 5 KB is FreeRTOS's number for tasks +/// that call `printf`; these bodies do not, and the honest way to size the pool is a painted-stack +/// watermark on the die, not this constant. +pub const thread_count = 7; +pub const requested_stack_bytes = 5 * 1024; + +fn hostedThreadCreate( + tname: [*:0]const u8, + tprio: u32, + tstack_size: u32, + start_routine: StartRoutine, + sr_arg: ?*anyopaque, +) callconv(.c) ?*anyopaque { + assert(state.installed); + // Priority is meaningless on a cooperative scheduler: a task runs until it blocks, and + // ESP-Hosted gives all seven the same priority anyway (RPC_TASK_PRIO and DFLT_TASK_PRIO are + // both 23, port_esp_hosted_host_os.h:65-68). + _ = tprio; + _ = tstack_size; + return @ptrCast(os.Thread.create(currentIo(), state.gpa, tname, start_routine, sr_arg)); +} + +fn hostedThreadCancel(handle: ?*anyopaque) callconv(.c) c_int { + const t: *os.Thread = @ptrCast(@alignCast(handle orelse return ret.invalid)); + return t.cancel(currentIo(), state.gpa); +} + +fn hostedThreadYield() callconv(.c) void { + // A zero-duration sleep is the portable yield, and on this runtime it is a documented one + // trip round the run queue rather than a no-op. Cancelation is swallowed because the C caller + // (`spi_hd_drv.c:568`, the only one in the tree) has nowhere to report it. + currentIo().sleep(.zero, os.clock) catch {}; +} + +// ============================================================================ 4. time + +fn hostedMsleep(mseconds: c_uint) callconv(.c) c_uint { + currentIo().sleep(.fromMilliseconds(mseconds), os.clock) catch {}; + return 0; +} + +fn hostedUsleep(useconds: c_uint) callconv(.c) c_uint { + currentIo().sleep(.fromMicroseconds(useconds), os.clock) catch {}; + return 0; +} + +/// Counted, because in this build every call is one turn of an ESP-Hosted RPC thread's not-ready +/// spin. See `State.hosted_sleep_calls`. +fn hostedSleep(seconds: c_uint) callconv(.c) c_uint { + state.hosted_sleep_calls += 1; + return hostedMsleep(seconds *| 1000); +} + +/// `_h_blocking_delay` is documented in ESP-Hosted as a "non sleepable delay - BLOCKING dead wait" +/// and implemented as `for (idx = 0; idx < 100*number; idx++)` on a `volatile` +/// (`port_esp_hosted_host_os.c:261-267`). That is a loop count, not a duration, and its wall-clock +/// meaning depends on the compiler and the CPU clock. +/// +/// It is reproduced as a real busy-wait rather than a sleep, because a caller reaching for this +/// specifically wants not to yield - and reproduced against `hal.systimer` rather than a loop +/// count, so the delay is at least defined. ESP-IDF's version at 360 MHz takes roughly 0.3 us per +/// unit; at this board's measured 90 MHz it would be about 1.1 us, and 1 us is the round number in +/// range. **Nothing in the tree calls this**, verified by grep, so no behaviour depends on the +/// choice. +/// +/// On a cooperative scheduler this starves every other task for the duration. That is inherent to +/// what the entry means, not a defect of this implementation. +fn hostedBlockingDelay(number: c_uint) callconv(.c) c_uint { + hal.systimer.delayMicros(number); + return 0; +} + +fn hostedGetTimeMs() callconv(.c) u64 { + return os.nowMs(currentIo()); +} + +// ============================================================================ timers + +/// A timer handle as C sees it. ESP-IDF hands back a heap pointer +/// (`port_esp_hosted_host_os.c:697`); this hands back a pointer to one, so `_h_timer_stop` can find +/// the slot and free the handle exactly as ESP-IDF's does. +const TimerHandle = struct { + slot: usize, +}; + +fn hostedTimerStart( + name: [*:0]const u8, + duration_ms: c_int, + kind: c_int, + handler: TimerHandler, + arg: ?*anyopaque, +) callconv(.c) ?*anyopaque { + assert(state.installed); + if (duration_ms < 0) return null; + const k: os.TimerKind = switch (kind) { + 0 => .oneshot, + 1 => .periodic, + else => { + // ESP-IDF logs "Unsupported timer type" and returns NULL (:720-725). + note("MARK PORT_TIMER_BAD_TYPE %s %d\r\n", .{ name, kind }); + return null; + }, + }; + const slot = state.timers.arm(currentIo(), @intCast(duration_ms), k, handler, arg) orelse { + note("MARK PORT_TIMER_SLOTS_FULL %s\r\n", .{name}); + return null; + }; + const h = state.gpa.create(TimerHandle) catch { + _ = state.timers.disarm(currentIo(), slot); + return null; + }; + h.* = .{ .slot = slot }; + return @ptrCast(h); +} + +fn hostedTimerStop(handle: ?*anyopaque) callconv(.c) c_int { + const h: *TimerHandle = @ptrCast(@alignCast(handle orelse return ret.fail)); + const r = state.timers.disarm(currentIo(), h.slot); + state.gpa.destroy(h); + return r; +} + +// ============================================================================ 5. GPIO + +/// `H_GPIO_MODE_DEF_*`, `port_esp_hosted_host_os.h:71-73`: bit 0 input, bit 1 output, bit 2 +/// open-drain. +const gpio_mode_input: u32 = 1 << 0; +const gpio_mode_output: u32 = 1 << 1; +const gpio_mode_open_drain: u32 = 1 << 2; + +/// `H_GPIO_PULL_UP` is 1 and `H_GPIO_PULL_DOWN` is 0 (`port_esp_hosted_host_os.h:83-84`) - note +/// that this is a *direction* selector and not a boolean, and the separate `enable` argument says +/// whether to turn that resistor on or off. +const gpio_pull_up: u32 = 1; + +/// `_h_config_gpio`. The `gpio_port` argument is always `H_GPIO_PORT_DEFAULT` / NULL on this chip +/// (`port_esp_hosted_host_config.h:435`); ESP-IDF ignores it too. +/// +/// ESP-IDF's version goes through `gpio_config`, which also clears both pulls +/// (`port_esp_hosted_host_os.c:746-758`). Reproduced, because the reset pin depends on it: GPIO54 +/// has an external pull-up and an internal pull-down fighting it would be a weak, marginal high. +fn hostedConfigGpio(gpio_port: ?*anyopaque, gpio_num: u32, mode: u32) callconv(.c) c_int { + _ = gpio_port; + if (gpio_num > hal.gpio.max_pin) return ret.invalid; + const pin: u8 = @intCast(gpio_num); + + hal.gpio.setFunction(pin, .gpio); + hal.gpio.setPull(pin, .none); + hal.gpio.setOpenDrain(pin, mode & gpio_mode_open_drain != 0); + hal.gpio.setInputEnable(pin, mode & gpio_mode_input != 0); + if (mode & gpio_mode_output != 0) { + // Point the matrix at the GPIO peripheral before enabling the driver, so the pad never + // spends an instant driven by whatever signal the matrix happened to hold. + hal.gpio.matrixOut(pin, hal.gpio.matrix_gpio_signal); + hal.gpio.outputEnable(pin); + } else { + hal.gpio.outputDisable(pin); + } + return ret.ok; +} + +fn hostedReadGpio(gpio_port: ?*anyopaque, gpio_num: u32) callconv(.c) c_int { + _ = gpio_port; + if (gpio_num > hal.gpio.max_pin) return ret.invalid; + return hal.gpio.getLevel(@intCast(gpio_num)); +} + +fn hostedWriteGpio(gpio_port: ?*anyopaque, gpio_num: u32, value: u32) callconv(.c) c_int { + _ = gpio_port; + if (gpio_num > hal.gpio.max_pin) return ret.invalid; + hal.gpio.setLevel(@intCast(gpio_num), if (value != 0) 1 else 0); + return ret.ok; +} + +/// `_h_pull_gpio(port, pin, pull_value, enable)`. +/// +/// The four-argument shape does not map onto one register field: the P4 has one pull-up bit and one +/// pull-down bit, and `hal.gpio.setPull` writes both in one store precisely so a pad can never end +/// up with two resistors fighting. Disabling one pull therefore means "leave the *other* alone", +/// which is read back rather than assumed. +fn hostedPullGpio(gpio_port: ?*anyopaque, gpio_num: u32, pull_value: u32, enable: u32) callconv(.c) c_int { + _ = gpio_port; + if (gpio_num > hal.gpio.max_pin) return ret.invalid; + const pin: u8 = @intCast(gpio_num); + const up = pull_value == gpio_pull_up; + if (enable != 0) { + hal.gpio.setPull(pin, if (up) .up else .down); + } else { + // gpio_pullup_dis / gpio_pulldown_dis clear one bit only. If the other pull is not set + // either, the pad ends up floating, which is what ESP-IDF leaves behind too. + const current = hal.gpio.getPull(pin); + const target: hal.gpio.Pull = if (up) + (if (current == .down) .down else .none) + else + (if (current == .up) .up else .none); + hal.gpio.setPull(pin, target); + } + return ret.ok; +} + +/// `_h_hold_gpio`. ESP-IDF calls `gpio_hold_en`, which latches a pad's output through a sleep or a +/// domain power-down so the slave is not reset by the host napping. +/// +/// This image never sleeps and never powers a domain down: `_h_config_host_power_save_hal_impl` and +/// `_h_start_host_power_save_hal_impl` are both loud stubs, and the only callers of this entry are +/// in `power_save_drv.c:210,230`, which those stubs make unreachable. Holding a pad against a sleep +/// that cannot happen is not a no-op worth pretending to - the P4's hold bit lives in +/// `LP_AON`/`HP_SYS` registers the HAL does not model, and writing them blind is how a pad gets +/// stuck. So this reports failure loudly instead. +fn hostedHoldGpio(gpio_port: ?*anyopaque, gpio_num: u32, hold_value: u32) callconv(.c) c_int { + _ = gpio_port; + state.stub_calls += 1; + note("MARK PORT_STUB _h_hold_gpio pin=%u hold=%u (no sleep support; nothing should reach this)\r\n", .{ gpio_num, hold_value }); + return ret.fail; +} + +/// `H_GPIO_INTR_*`, `port_esp_hosted_host_config.h:56-62`. The values coincide exactly with the +/// P4's `GPIO_PINn_INT_TYPE` encoding (`gpio_reg.h:377-381`), which is not a coincidence: the +/// enum was written from it. +fn intrTypeFromHosted(intr_type: u32) ?hal.gpio.IntrType { + return switch (intr_type) { + 0 => .disable, + 1 => .posedge, + 2 => .negedge, + 3 => .anyedge, + 4 => .low_level, + 5 => .high_level, + else => null, + }; +} + +/// `_h_config_gpio_as_interrupt`. +/// +/// ESP-IDF's version (`port_esp_hosted_host_os.c:760-797`) configures the pad as an input with a +/// pull that opposes the edge being detected, installs IDF's shared GPIO ISR service, adds a +/// per-pin handler, then sets the trigger type and enables. Same five steps here, with `hal.gpio` +/// and `hal.intr` in place of the driver: +/// +/// 1. pad as input, pull opposing the edge - a floating pad on an edge-triggered interrupt is a +/// free-running interrupt source. +/// 2. record (pin, handler, arg) in `state.gpio_isrs`. +/// 3. arm the pad on GPIO interrupt line 0, which is the line ESP-IDF uses. +/// 4. route `gpio_intr0` to a CLIC line and give it `gpioDispatch`, once. +/// 5. enable. +/// +/// The CLIC trigger is **level**, not edge: the GPIO peripheral holds its line asserted while any +/// status bit is set, and the handler clears the status. An edge-triggered CLIC line here would +/// lose a second pad's event that arrived while the first was being serviced. +/// +/// Nothing in the SDIO transport calls this. Its callers are `spi_drv.c:625,628`, +/// `spi_hd_drv.c:548` and `power_save_drv.c:68`. It is implemented rather than stubbed because it +/// costs little and because a host-wakeup pin is the obvious next use. +fn hostedConfigGpioAsInterrupt( + gpio_port: ?*anyopaque, + gpio_num: u32, + intr_type: u32, + handler: IsrHandler, + arg: ?*anyopaque, +) callconv(.c) c_int { + _ = gpio_port; + if (gpio_num > hal.gpio.max_pin) return ret.invalid; + const pin: u8 = @intCast(gpio_num); + const t = intrTypeFromHosted(intr_type) orelse { + note("MARK PORT_GPIO_BAD_INTR_TYPE %u\r\n", .{intr_type}); + return ret.invalid; + }; + + // ESP-IDF pulls up for a falling edge and down for anything else (:771-775). + hal.gpio.configureInput(pin, .{ .pull = if (t == .negedge) .up else .down }); + + const slot = blk: { + for (&state.gpio_isrs) |*s| if (s.pin == pin) break :blk s; + for (&state.gpio_isrs) |*s| if (s.handler == null) break :blk s; + note("MARK PORT_GPIO_ISR_SLOTS_FULL pin=%u\r\n", .{gpio_num}); + return ret.fail; + }; + slot.* = .{ .pin = pin, .handler = handler, .arg = arg }; + + if (!gpio_line_attached) { + gpio_line_attached = true; + // mtvec, MTVT, the threshold and MIE, before a line that `configureLine` enables as its + // last act can be delivered anywhere. See `takeInterruptControl`. + takeInterruptControl(); + hal.intr.routeId(@intFromEnum(hal.intr.Source.gpio_intr0), config.gpio_clic_line); + hal.intr.configureLine(config.gpio_clic_line, .{ + .handler = gpioDispatch, + .trigger = .level, + }); + } + hal.gpio.setInterrupt(pin, t, .line0); + return ret.ok; +} + +fn hostedTeardownGpioInterrupt(gpio_port: ?*anyopaque, gpio_num: u32) callconv(.c) c_int { + _ = gpio_port; + if (gpio_num > hal.gpio.max_pin) return ret.invalid; + const pin: u8 = @intCast(gpio_num); + hal.gpio.disableInterrupt(pin); + hal.gpio.clearInterrupt(pin); + for (&state.gpio_isrs) |*s| { + if (s.pin == pin) s.* = .{}; + } + return ret.ok; +} + +var gpio_line_attached: bool = false; + +/// The one CLIC handler behind every registered pad. Reads the whole pending mask once, clears it +/// once, then dispatches - so an event on a second pad arriving mid-dispatch is caught by the next +/// interrupt rather than lost. +/// +/// The status is cleared *before* the handlers run. For an edge-triggered pad that is the correct +/// order: clearing after the handler would drop an edge that arrived during it. +fn gpioDispatch(line: u5) void { + _ = line; + const pending = hal.gpio.pendingMask(.line0); + hal.gpio.clearInterrupts(pending.low, pending.high); + for (&state.gpio_isrs) |*s| { + const h = s.handler orelse continue; + const bit: u32 = @as(u32, 1) << @intCast(if (s.pin < 32) s.pin else s.pin - 32); + const hit = if (s.pin < 32) pending.low & bit else pending.high & bit; + if (hit != 0) h(s.arg); + } +} + +// ============================================================================ 6. SDIO + +/// `ESP_ADDRESS_MASK`, `host/drivers/transport/sdio/sdio_reg.h:87`. Slave scratch registers live in +/// the low 10 bits of function 1's address space, and ESP-Hosted masks every register address with +/// this before the transfer (`port_esp_hosted_host_sdio.c:500,523`). Block transfers are *not* +/// masked, which is why `ESP_SLAVE_CMD53_END_ADDR - data_left` works. +const esp_address_mask: u32 = 0x3FF; +/// `ESP_BLOCK_SIZE`, `sdio_reg.h:39`. +const esp_block_size: u32 = 512; +/// The SDIO function ESP-Hosted talks to. `SDIO_FUNC_1`. +const sdio_func: u3 = 1; + +/// `ESP_OK` / `ESP_FAIL` as `esp_err_t`, which is what the `_h_sdio_*` entries return and what +/// `sdio_drv.c` tests against zero. +const esp_ok: c_int = 0; +const esp_fail: c_int = -1; + +fn busCtx(ctx: ?*anyopaque) ?*BusContext { + const p = ctx orelse return null; + const b: *BusContext = @ptrCast(@alignCast(p)); + // ESP-IDF returns a pointer to one file-static context; anything else is a bug, and a wild + // pointer here would be a wild bus. + if (b != &state.bus) return null; + return b; +} + +/// `_h_bus_init` = `hosted_sdio_init` (`port_esp_hosted_host_sdio.c:317-399`): bring the SDMMC host +/// and slot up, create the bus mutex, return the context. Guarded against a second call, as the +/// original is (`:322-326`). +/// +/// The slot, width and clock are `hal.sdmmc`'s defaults, which are this board's measured working +/// configuration: slot 1, 4-bit, 40 MHz, CLK 18 / CMD 19 / D0-D3 14-17. +fn hostedBusInit() callconv(.c) ?*anyopaque { + assert(state.installed); + if (state.bus.up) { + note("MARK PORT_SDIO_ALREADY_UP\r\n", .{}); + return @ptrCast(&state.bus); + } + hal.sdmmc.init(.{}) catch |e| { + note("MARK PORT_SDIO_INIT_FAIL %s\r\n", .{@errorName(e).ptr}); + return null; + }; + state.bus = .{ .lock = .{}, .up = true }; + return @ptrCast(&state.bus); +} + +fn hostedBusDeinit(ctx: ?*anyopaque) callconv(.c) c_int { + const b = busCtx(ctx) orelse return esp_fail; + b.up = false; + return esp_ok; +} + +/// `_h_sdio_card_init` = `hosted_sdio_card_init` + `hosted_sdio_card_fn_init` +/// (`port_esp_hosted_host_sdio.c:141-217, 401-471`). +/// +/// `hal.sdmmc.cardInit` does the SD/SDIO card identification and programmes the host's block size. +/// What is left is the part that is ESP-Hosted's protocol rather than the bus's: enable function 1, +/// wait for it to report ready, enable its interrupt, and set the CCCR block size for functions 0 +/// and 1. Those writes are idempotent and the read-back is the check; the sequence is reproduced +/// in ESP-IDF's order because that order is what this board was observed to come up with. +/// +/// Failure returns `ESP_FAIL` rather than asserting, because the caller retries: `sdio_drv.c:1638` +/// loops up to `CARD_INIT_TIMEOUT_MS`, and the first register reads after a reset legitimately +/// fail while the C6 is still booting (`:150-153`). +fn hostedSdioCardInit(ctx: ?*anyopaque, show_config: bool) callconv(.c) c_int { + const b = busCtx(ctx) orelse return esp_fail; + _ = b; + hal.sdmmc.cardInit() catch |e| { + note("MARK PORT_SDIO_CARD_INIT_FAIL %s\r\n", .{@errorName(e).ptr}); + return esp_fail; + }; + if (show_config) { + note("MARK PORT_SDIO slot=1 width=4 khz=40000 clk=18 cmd=19 d0-3=14,15,16,17 reset=%u\r\n", .{ + @as(u32, config.reset_pin), + }); + } + return sdioFunctionInit(); +} + +// CCCR and FBR offsets, `esp-idf/components/sdmmc/include/sd_protocol_defs.h:511-533`. +const cccr_fn_enable: u17 = 0x02; +const cccr_fn_ready: u17 = 0x03; +const cccr_int_enable: u17 = 0x04; +const cccr_bus_width: u17 = 0x07; +const cccr_blksize_l: u17 = 0x10; +const cccr_blksize_h: u17 = 0x11; +const fbr_start: u17 = 0x100; +/// `FUNC1_EN_MASK`, `port_esp_hosted_host_sdio.c:29`. +const func1_en_mask: u8 = 1 << 1; +/// `SDIO_INIT_MAX_RETRY`, `:30`. +const sdio_init_max_retry = 10; + +fn sdioFunctionInit() c_int { + // Function 0 is the CCCR; every access here is CMD52 on function 0. + var ioe = cmd52(0, cccr_fn_enable) orelse return esp_fail; + cmd52w(0, cccr_fn_enable, ioe | func1_en_mask) orelse return esp_fail; + + // Poll IOR until function 1 reports ready. 10 tries, 10 ms apart (:180-192). + var tries: u32 = 0; + while (tries < sdio_init_max_retry) : (tries += 1) { + const ior = cmd52(0, cccr_fn_ready) orelse return esp_fail; + if (ior & func1_en_mask != 0) break; + _ = hostedMsleep(10); + } + if (tries >= sdio_init_max_retry) { + note("MARK PORT_SDIO_FN1_NOT_READY\r\n", .{}); + return esp_fail; + } + + // Master interrupt enable (bit 0) plus function 1's own (:196-198). + const ie = cmd52(0, cccr_int_enable) orelse return esp_fail; + cmd52w(0, cccr_int_enable, ie | 1 | func1_en_mask) orelse return esp_fail; + + const bus_width = cmd52(0, cccr_bus_width) orelse return esp_fail; + + // CCCR block size for function 0, then function 1 through its FBR (:120-137, 208-214). + if (setBlockSize(0, esp_block_size) != esp_ok) return esp_fail; + if (setBlockSize(1, esp_block_size) != esp_ok) return esp_fail; + + ioe = cmd52(0, cccr_fn_enable) orelse return esp_fail; + note("MARK PORT_SDIO_FN1 ioe=0x%02x ie=0x%02x bus_width=0x%02x\r\n", .{ + @as(u32, ioe), @as(u32, ie | 1 | func1_en_mask), @as(u32, bus_width), + }); + return esp_ok; +} + +fn setBlockSize(func: u3, value: u16) c_int { + const offset: u17 = fbr_start * @as(u17, func); + const lo: u8 = @truncate(value); + const hi: u8 = @truncate(value >> 8); + cmd52w(0, offset + cccr_blksize_l, lo) orelse return esp_fail; + cmd52w(0, offset + cccr_blksize_h, hi) orelse return esp_fail; + const rb_lo = cmd52(0, offset + cccr_blksize_l) orelse return esp_fail; + const rb_hi = cmd52(0, offset + cccr_blksize_h) orelse return esp_fail; + const rb = @as(u16, rb_hi) << 8 | rb_lo; + return if (rb == value) esp_ok else esp_fail; +} + +fn cmd52(func: u3, addr: u17) ?u8 { + return hal.sdmmc.cmd52Read(func, addr) catch null; +} + +fn cmd52w(func: u3, addr: u17, value: u8) ?void { + hal.sdmmc.cmd52Write(func, addr, value) catch return null; + return {}; +} + +/// `_h_sdio_card_deinit` frees IDF's DMA bounce buffer (`port_esp_hosted_host_sdio.c:473-487`). +/// `hal.sdmmc` owns its bounce buffer statically, so there is nothing to free. +fn hostedSdioCardDeinit(ctx: ?*anyopaque) callconv(.c) c_int { + _ = busCtx(ctx) orelse return esp_fail; + return esp_ok; +} + +/// `lock_required` exists because ESP-IDF's SDMMC driver is shared: `sdio_drv.c` reaches the bus +/// from four tasks, and a CMD53 that interleaves with another CMD53 is a corrupt transfer. Some +/// call sites already hold the bus lock (`SDIO_DRV_LOCK`) and pass false to avoid taking it twice; +/// the rest pass true. +/// +/// **It is still required here**, and this is the one place where a cooperative scheduler does not +/// let a lock go. Cooperative means no task is preempted between two *instructions*; it does not +/// mean a task cannot yield in the middle of a transfer, and `hal.sdmmc`'s CMD53 path does exactly +/// that if it waits on the SDMMC host's interrupt. A second task entering `cmd53Read` while the +/// first is parked inside one would reprogramme the descriptor under it. The lock is what makes +/// "one transfer at a time" true, and it is cheap: `Io.Mutex.tryLock` is one compare-exchange when +/// uncontended, which is every call on the fast path. +fn sdioLock(b: *BusContext, required: bool) void { + if (required) _ = b.lock.lock(state.io, .forever); +} + +fn sdioUnlock(b: *BusContext, required: bool) void { + if (required) _ = b.lock.unlock(state.io); +} + +/// `_h_sdio_read_reg`: function 1, address masked, CMD52 for one byte and CMD53 byte mode with an +/// incrementing address for more (`port_esp_hosted_host_sdio.c:489-511`). +fn hostedSdioReadReg(ctx: ?*anyopaque, reg: u32, data: [*]u8, size: u16, lock_required: bool) callconv(.c) c_int { + const b = busCtx(ctx) orelse return esp_fail; + const addr: u17 = @intCast(reg & esp_address_mask); + sdioLock(b, lock_required); + defer sdioUnlock(b, lock_required); + if (size <= 1) { + data[0] = hal.sdmmc.cmd52Read(sdio_func, addr) catch return esp_fail; + return esp_ok; + } + hal.sdmmc.cmd53Read(sdio_func, addr, data[0..size], true) catch return esp_fail; + return esp_ok; +} + +fn hostedSdioWriteReg(ctx: ?*anyopaque, reg: u32, data: [*]u8, size: u16, lock_required: bool) callconv(.c) c_int { + const b = busCtx(ctx) orelse return esp_fail; + const addr: u17 = @intCast(reg & esp_address_mask); + sdioLock(b, lock_required); + defer sdioUnlock(b, lock_required); + if (size <= 1) { + hal.sdmmc.cmd52Write(sdio_func, addr, data[0]) catch return esp_fail; + return esp_ok; + } + hal.sdmmc.cmd53Write(sdio_func, addr, data[0..size], true) catch return esp_fail; + return esp_ok; +} + +/// `_h_sdio_read_block` / `_h_sdio_write_block`, `port_esp_hosted_host_sdio.c:536-576`, with the +/// splitting from `sdio_read_fromio`/`sdio_write_toio` (`:221-292`): +/// +/// * the length is first rounded **up** to a multiple of four (`H_SDIO_TX_LEN_TO_TRANSFER`, +/// `port_esp_hosted_host_config.h:274-275`), because the slave's FIFO is word-wide; +/// * while 512 bytes or more remain, transfer whole 512-byte blocks; +/// * transfer the remainder in byte mode; +/// * the address advances by every chunk, and is **not** masked - block transfers address the +/// slave's data window, not its scratch registers. +/// +/// Rounding up means reading or writing past `size`. That is ESP-Hosted's design, not an accident: +/// its buffers come from `_h_malloc_align(len, 64)`, so there are always at least 64 usable bytes +/// at the end - and this port's `_h_malloc_align` rounds the *allocation* up to the alignment for +/// exactly this reason. A caller that hands a tightly-sized buffer to a block transfer would have +/// the same bug under ESP-IDF. +fn hostedSdioReadBlock(ctx: ?*anyopaque, reg: u32, data: [*]u8, size: u16, lock_required: bool) callconv(.c) c_int { + const b = busCtx(ctx) orelse return esp_fail; + sdioLock(b, lock_required); + defer sdioUnlock(b, lock_required); + if (size <= 1) { + // Unmasked, unlike the `_reg` entries: `hosted_sdio_read_block` has no + // `reg &= ESP_ADDRESS_MASK` (port_esp_hosted_host_sdio.c:536-555). Masking here would + // fold `ESP_SLAVE_CMD53_END_ADDR - data_left` (sdio_drv.c:756) onto a scratch register. + data[0] = hal.sdmmc.cmd52Read(sdio_func, @intCast(reg)) catch return esp_fail; + return esp_ok; + } + return blockTransfer(.read, reg, data, size); +} + +fn hostedSdioWriteBlock(ctx: ?*anyopaque, reg: u32, data: [*]u8, size: u16, lock_required: bool) callconv(.c) c_int { + const b = busCtx(ctx) orelse return esp_fail; + sdioLock(b, lock_required); + defer sdioUnlock(b, lock_required); + if (size <= 1) { + // Unmasked; see `hostedSdioReadBlock`. + hal.sdmmc.cmd52Write(sdio_func, @intCast(reg), data[0]) catch return esp_fail; + return esp_ok; + } + return blockTransfer(.write, reg, data, size); +} + +fn blockTransfer(comptime dir: enum { read, write }, reg: u32, data: [*]u8, size: u16) c_int { + // H_SDIO_{TX,RX}_LEN_TO_TRANSFER: (x + 3) & ~3. + const total: u32 = (@as(u32, size) + 3) & ~@as(u32, 3); + var remaining: u32 = total; + var addr: u32 = reg; + var at: u32 = 0; + + while (remaining >= esp_block_size) { + // H_SDIO_{TX,RX}_BLOCKS_TO_TRANSFER: all whole blocks in one command unless the build + // forces one block at a time (port_esp_hosted_host_config.h:297-308). + const chunk = (remaining / esp_block_size) * esp_block_size; + const slice = data[at .. at + chunk]; + switch (dir) { + .read => hal.sdmmc.cmd53Read(sdio_func, @intCast(addr), slice, true) catch return esp_fail, + .write => hal.sdmmc.cmd53Write(sdio_func, @intCast(addr), slice, true) catch return esp_fail, + } + remaining -= chunk; + at += chunk; + addr += chunk; + } + if (remaining > 0) { + const slice = data[at .. at + remaining]; + switch (dir) { + .read => hal.sdmmc.cmd53Read(sdio_func, @intCast(addr), slice, true) catch return esp_fail, + .write => hal.sdmmc.cmd53Write(sdio_func, @intCast(addr), slice, true) catch return esp_fail, + } + } + return esp_ok; +} + +/// `_h_sdio_wait_slave_intr`: block until the C6 asserts its SDIO interrupt on D1. +/// +/// The arming order is IDF's, from `sd_host_sdmmc.c:396-426`: mask the card interrupt, drop the +/// previous wake's latch, look once at what is pending, and only then unmask and sleep. The look +/// is not optional - the capture is negedge-triggered, so an edge that arrived while this task was +/// awake is not going to arrive again. +/// +/// ### The storm this function used to cause +/// +/// Measured on the die: the first call here killed the machine. Every task starved, including one +/// that does nothing but sleep and print a heartbeat, from the instant `configureLine` set the +/// line's IE bit. On a cooperative scheduler nothing that *blocks* can do that. It was an +/// interrupt storm. +/// +/// The controller drives a single line into the CLIC and asserts it whenever `RINTSTS & INTMASK` +/// (or the IDMAC's `IDSTS & IDINTEN`) is non-zero - not just for the card interrupt this function +/// waits on. Two separate causes were holding it high permanently: `INTMASK` carried +/// `Event.default`, whose card-detect bit no command path ever clears, and `initDma` had unmasked +/// the IDMAC's three completion interrupts with nothing ever clearing `IDSTS` after a transfer. +/// Either one is enough. +/// +/// A level-triggered line whose source is still asserting re-enters the moment the handler +/// `mret`s. The old `sdioDispatch` tested `slaveInterruptPending()` *first* and took an early +/// return when the cause was not the card interrupt - without masking or clearing anything. So +/// the line stayed high, the core re-entered, and it never came back. `hal.intr`'s module comment +/// describes this precise failure for lines the ROM left armed (`intr.zig:512-518`); this was the +/// same bug, self-inflicted. +/// +/// Three invariants fix it, none of which depends on guessing which bit was set: +/// +/// * **the handler deasserts on every path**, before it reads anything at all; +/// * **only this function arms.** `hal.intr.configureLine` enables the line as its last act, +/// which is exactly what must not happen at configuration time, so the line is configured +/// with the individual setters and left disabled; +/// * **the controller is silent unless armed** - `hal.sdmmc`'s half of the fix, which reduces +/// the set of possible causes to one. +/// +/// ### Level, not edge, and why the answer is not "either works" +/// +/// Two different trigger behaviours meet on this path, and conflating them sends you tuning the +/// wrong knob. **Card to controller is an edge**: D1's negedge is captured once into RINTSTS, +/// which is why step 3 below reads D1's *pad* rather than the latch before sleeping. +/// **Controller to CLIC is a level**: RINTSTS is a sticky write-1-to-clear latch and MINTSTS is +/// `RINTSTS & INTMASK`, so the controller's single output stays asserted until software masks or +/// clears the bit that raised it. The CLIC trigger describes that second stage and only that one, +/// so it is `.level`. +/// +/// `.edge` would be wrong three times over, and the third is the one that bites. It would need an +/// `edgeAck` this handler does not do. It would drop a re-assert that arrived while the line was +/// still high, because there is no second rising edge to capture. And it would *hide* a handler +/// that fails to deassert - the re-entry would stop, the storm would go away, and the bug would +/// still be there, waiting for the day something else holds MINTSTS non-zero. A level trigger +/// makes that failure loud and local, which is worth more than a trigger type that works by luck. +/// +/// ### The precondition that was missing, and was read as a mask that would not stick +/// +/// The line was configured, routed and armed - and nothing in this image had taken ownership of +/// the interrupt controller. `takeInterruptControl` is that step and its comment has the detail; +/// the short form is that `hal.intr.setHandler` files a handler in a table the core does not +/// consult until `hal.intr.init()` has written mtvec and MTVT, and that the threshold and +/// mstatus.MIE are equally this image's job and were nobody's. Neither diagnostic that reported +/// `intmask=0` could have shown anything else, because both read INTMASK after a deliberate +/// disarm; `MARK PORT_SDIO_ARM` carries the read-back that can. +/// +/// `ticks_to_wait` is FreeRTOS ticks. The only caller (`sdio_drv.c:1191`) passes +/// `HOSTED_BLOCK_MAX`, so the bounded branch exists for completeness; at ESP-Hosted's recommended +/// tick rate one tick is one millisecond. +fn hostedSdioWaitSlaveIntr(ctx: ?*anyopaque, ticks_to_wait: u32) callconv(.c) c_int { + if (busCtx(ctx) == null) return esp_fail; + + // One unconditional trip round the run queue, before anything else. + // + // Every other path out of this function can return without ever having slept: the pad read at + // step 3, the latch read after it, and `sdioPoll`'s fast path all answer "yes, now". That is + // correct - and it means a card holding D1 low that the C declines to drain (no NEW_PACKET + // bit, `sdio_drv.c:1247-1251`) turns `sdio_read_task`'s `for (;;)` into a loop with no + // yield in it anywhere, because the C has none of its own either. A blocking entry point that + // can return without blocking has to supply the scheduling point itself; the alternative is + // the same total starvation as the interrupt storm, reached by a different road. + state.io.sleep(.zero, os.clock) catch {}; + + // Configured off by default on this board: see `Config.sdio_use_interrupt`. Checked before the + // line is ever configured, so with polling selected the CLIC is not touched at all. + if (!config.sdio_use_interrupt) return sdioPoll(ticks_to_wait); + + // Enough foreign handler entries, or enough calls the interrupt failed to deliver, and this + // line is not usable on this board whatever the mask says. Poll instead: slower per look, but + // bounded, proven, and faster than a 20 ms re-look that is carrying the transport on its own. + if (state.sdio_intr_foreign >= sdio_foreign_limit) return sdioPoll(ticks_to_wait); + if (state.sdio_intr_missed >= sdio_missed_limit) return sdioPoll(ticks_to_wait); + + if (!sdio_line_configured) { + sdio_line_configured = true; + // First, and the step whose absence produced every LAPSE this board has reported: mtvec, + // MTVT, the threshold and mstatus.MIE. + takeInterruptControl(); + hal.intr.route(hal.sdmmc.interrupt_source, config.sdio_clic_line); + // `hal.intr.configureLine` in its documented order, minus the `setEnabled(line, true)` it + // finishes with. See the storm note: enabling here is the bug. + hal.intr.setHandler(config.sdio_clic_line, sdioDispatch); + hal.intr.setTrigger(config.sdio_clic_line, .level); + hal.intr.setPriority(config.sdio_clic_line, sdio_clic_priority); + hal.intr.setVectored(config.sdio_clic_line, false); + hal.intr.setEnabled(config.sdio_clic_line, false); + + // The whole delivery chain above the controller, once, before the first sleep. Each field + // is a distinct way for the line to exist and never arrive, and each has a different fix: + // `routed=99` is a matrix write that missed, `routed` unequal to `line` is two owners of + // one line, `thresh >= prio` masks it however armed it is (the comparison is inclusive), + // `mie=0` masks everything, and `mtvec` unequal to `want_mtvec` means the handler the core + // would reach is not this image's. + note("MARK PORT_SDIO_CLIC line=%u source=%u routed=%u prio=%u trig=%u thresh=%u mie=%u mtvec=0x%08x want_mtvec=0x%08x\r\n", .{ + @as(u32, config.sdio_clic_line), + @as(u32, @intFromEnum(hal.sdmmc.interrupt_source)), + @as(u32, hal.intr.routedLine(hal.sdmmc.interrupt_source) orelse 99), + @as(u32, hal.intr.getPriority(config.sdio_clic_line)), + @as(u32, @intFromEnum(hal.intr.getTrigger(config.sdio_clic_line))), + @as(u32, hal.intr.getThreshold()), + @as(u32, @intFromBool(hal.intr.globalEnabled())), + hal.intr.readMtvec(), + hal.intr.trapEntryAddress() | hal.intr.mtvec_mode_clic, + }); + } + + // A bounded wait that loops, rather than the unbounded one the caller asked for. + // + // The lost-edge case that used to need this is now handled properly at step 3 of the arming + // sequence below, so this is no longer the mechanism - it is the net under it. It stays + // because an unbounded futex wait is precisely the shape of failure that cost an afternoon: + // silent, indistinguishable from a card that never called, and impossible to report on. A + // 20 ms re-look turns "the radio is dead" into `MARK PORT_SDIO_LAPSE` with the registers + // attached, and costs that latency only on beats where the interrupt did not arrive. + // + // `sdio_drv.c:1188` is right that a finite wait is unusable *for the caller*, so the loop, not + // the wait, is what honours `HOSTED_BLOCK_MAX`: this function still only returns when there is + // something to report. The property gained is that no path through it can be silent for ever. + const bounded = ticks_to_wait != std.math.maxInt(u32); + const deadline = os.nowMs(state.io) + ticks_to_wait; + + while (true) { + // Read before arming, so an interrupt taken between here and the futex wait cannot be + // lost: `futexWaitTimeout` returns immediately on a value that no longer matches. + const seen = state.sdio_intr_epoch.load(.acquire); + + // Steps 1-4 of `sd_host_sdmmc.c:404-426`, in that order, as written out on + // `hal.sdmmc.setSlaveInterruptEnabled`. Getting the order wrong loses wakeups; getting + // step 3 wrong loses them permanently. + hal.sdmmc.setSlaveInterruptEnabled(false); + hal.sdmmc.clearSlaveInterrupt(); + + // Step 3, and the one that cannot be done with the controller's registers alone. RINTSTS + // is a latch: it says "a negedge was captured", and step 2 has just thrown that away. D1's + // pad is a level: it says "the card is holding the line low *now*". A C6 that is still + // waiting to be drained is exactly the second without the first, and sleeping on it waits + // for an edge that has already happened. The latch is tested too, for the window between + // the clear above and this read. + // + // This is not a window that lapsed - nothing has been armed and nothing has slept - so it + // leaves `sdio_intr_missed` alone. + if (hal.sdmmc.slaveInterruptAsserted() or hal.sdmmc.slaveInterruptPending()) return esp_ok; + + // Source first, CLIC last: the line must not be deliverable while the only cause it is + // allowed to have is still masked. The unmask reads INTMASK back inside its own masked + // region, which is the only reading of that register that can answer "did it stick". + const armed = hal.sdmmc.armSlaveInterrupt(); + state.sdio_armed_intmask = armed.intmask; + state.sdio_armed_mintsts = armed.mintsts; + hal.intr.setEnabled(config.sdio_clic_line, true); + + // `stuck=1` retires the "the unmask does not stick" hypothesis; `stuck=0` confirms it, with + // the word that was wanted printed beside the word the register returned. Budgeted, + // because it is a property of the configuration rather than of the beat. + sdioMark(&state.sdio_arm_marks, "MARK PORT_SDIO_ARM stuck=%u want=0x%08x intmask=0x%08x mintsts=0x%08x rintsts=0x%08x ie=%u\r\n", .{ + @as(u32, @intFromBool(armed.stuck())), + armed.want, + armed.intmask, + armed.mintsts, + armed.rintsts, + @as(u32, @intFromBool(hal.intr.isEnabled(config.sdio_clic_line))), + }); + + // Timeout and cancelation are indistinguishable here and neither is a result; the epoch is + // the only thing that says whether the handler ran. + state.io.futexWaitTimeout(u32, &state.sdio_intr_epoch.raw, seen, .{ + .duration = .{ .clock = os.clock, .raw = .fromMilliseconds(sdio_relook_ms) }, + }) catch {}; + + // The CLIC's own pending bit, read *before* the disarm, because it is the discriminator a + // lapse otherwise has no way to report: `pend=1` with no handler entry means the CLIC + // latched this line and the core never took it, so the fault is mtvec, the threshold or + // MIE rather than the controller or the C6. + const clic_pending = hal.intr.isPending(config.sdio_clic_line); + + // Idempotent: on a real wake the handler already did both. On a lapse it did not, and an + // armed line with nobody waiting is how a storm gets its second chance. + disarmSdioLine(); + + if (state.sdio_intr_epoch.load(.acquire) != seen) { + // The handler ran. It deliberately does not clear the latched SDIO bit - clearing it + // while D1 is still low would drop the next wakeup - so the bit still being set is + // what distinguishes "the C6 called" from "something else held the controller's line + // high and the handler is who noticed". + if (hal.sdmmc.slaveInterruptPending()) { + state.sdio_intr_foreign = 0; + state.sdio_intr_missed = 0; + // **The line that says the interrupt works.** Until now a successful delivery was + // the only outcome that printed nothing at all, so a console showing idle lapses + // and no wakes was indistinguishable from a console showing a dead CLIC - which is + // exactly the ambiguity that made the last flash inconclusive. `mintsts` is the + // word the controller's output follows, captured at handler entry before the + // disarm zeroed it; this slot's bit set in it is delivery proven end to end. + sdioMark(&state.sdio_wake_marks, "MARK PORT_SDIO_WAKE n=%u mintsts=0x%08x rintsts=0x%08x idsts=0x%08x\r\n", .{ + state.sdio_intr_epoch.load(.acquire), + state.sdio_intr_mintsts.load(.acquire), + state.sdio_intr_rintsts.load(.acquire), + state.sdio_intr_idsts.load(.acquire), + }); + return esp_ok; + } + state.sdio_intr_foreign += 1; + sdioMark(&state.sdio_foreign_marks, "MARK PORT_SDIO_FOREIGN n=%u mintsts=0x%08x rintsts=0x%08x idsts=0x%08x armed_intmask=0x%08x\r\n", .{ + state.sdio_intr_foreign, + state.sdio_intr_mintsts.load(.acquire), + state.sdio_intr_rintsts.load(.acquire), + state.sdio_intr_idsts.load(.acquire), + state.sdio_armed_intmask, + }); + if (state.sdio_intr_foreign >= sdio_foreign_limit) { + note("MARK PORT_SDIO_POLLING abandoning CLIC line %u\r\n", .{ + @as(u32, config.sdio_clic_line), + }); + return sdioPoll(ticks_to_wait); + } + } else { + // Nobody entered the handler. Ask both ends directly before calling it a lapse - the + // pad for a card asserting now, the latch for an edge captured while the CLIC was + // being taken down. + // + // This is the one reading that separates the two things a lapse can mean, and it is + // why `sdio_intr_lapses` alone is not a fault signal. **The card is calling and the + // interrupt did not deliver it**: a real frame has just paid up to `sdio_relook_ms` of + // latency, the re-look is doing the interrupt's job, and four of those in a row is a + // configuration that will not fix itself - so the line goes back to the poll, which is + // twenty times quicker at exactly this. + if (hal.sdmmc.slaveInterruptAsserted() or hal.sdmmc.slaveInterruptPending()) { + state.sdio_intr_missed += 1; + sdioMark(&state.sdio_missed_marks, "MARK PORT_SDIO_MISSED n=%u pend=%u armed_intmask=0x%08x armed_mintsts=0x%08x rintsts=0x%08x\r\n", .{ + state.sdio_intr_missed, + @as(u32, @intFromBool(clic_pending)), + state.sdio_armed_intmask, + state.sdio_armed_mintsts, + hal.sdmmc.interruptStatusRaw(), + }); + if (state.sdio_intr_missed >= sdio_missed_limit) { + note("MARK PORT_SDIO_POLLING abandoning CLIC line %u after %u undelivered calls\r\n", .{ + @as(u32, config.sdio_clic_line), + state.sdio_intr_missed, + }); + return sdioPoll(ticks_to_wait); + } + return esp_ok; + } + + // The other meaning: the C6 had nothing to say. Free, and the resting state of an idle + // link - which is the whole point of waiting on an interrupt instead of polling. + state.sdio_intr_lapses +%= 1; + // `armed_*` is what the mask was during the window that lapsed; `now_*` is the + // disarmed state. Both are printed so the two can no longer be mistaken for each + // other: `now_intmask=0` is expected here, and always was. + sdioMark(&state.sdio_lapse_marks, "MARK PORT_SDIO_LAPSE n=%u armed_intmask=0x%08x armed_mintsts=0x%08x pend=%u now_rintsts=0x%08x now_intmask=0x%08x\r\n", .{ + state.sdio_intr_lapses, + state.sdio_armed_intmask, + state.sdio_armed_mintsts, + @as(u32, @intFromBool(clic_pending)), + hal.sdmmc.interruptStatusRaw(), + hal.sdmmc.interruptMaskRaw(), + }); + } + + // The one diagnostic with no budget, because the ratio it reports is the whole question and + // it stays interesting after every other line has gone quiet. `wakes` is handler entries: + // rising with `lapses` means the interrupt is carrying the transport and the re-look is + // only covering the idle gaps, flat at zero means the CLIC is not delivering and the + // re-look is doing all of it. + state.sdio_windows +%= 1; + if (state.sdio_windows % sdio_tally_every == 0) { + note("MARK PORT_SDIO_TALLY windows=%u wakes=%u lapses=%u missed=%u foreign=%u\r\n", .{ + state.sdio_windows, + state.sdio_intr_epoch.load(.acquire), + state.sdio_intr_lapses, + state.sdio_intr_missed, + state.sdio_intr_foreign, + }); + } + + if (bounded and os.nowMs(state.io) >= deadline) return esp_fail; + } +} + +/// Consecutive foreign handler entries after which the interrupt is abandoned for polling. Four, +/// because one can be a race and four in a row is a configuration that will not fix itself. +const sdio_foreign_limit: u32 = 4; + +/// Consecutive undelivered calls - lapsed windows whose re-look found the card already asserting - +/// after which the interrupt is abandoned for polling. +/// +/// Four, for the same reason as `sdio_foreign_limit`: one can be a race against the CLIC being +/// taken down, four in a row is a configuration. At `sdio_relook_ms` each that is 80 ms of +/// degraded latency before the line is given up, well inside one of the transport's own 200 ms +/// retry turns (`transport_drv.c:233`). +/// +/// This bound is what makes flipping `sdio_use_interrupt` to `true` an experiment rather than a +/// bet. Without it, a board where delivery is still broken would give every received frame 20 ms +/// instead of the poll's 1 ms, for ever, with eight budgeted MARK lines to say so. Note that it +/// counts *undelivered calls* and not lapses: an idle card lapses every window by construction, +/// and penalising that would trade the interrupt away 160 ms after boot on a link that was +/// working perfectly. +const sdio_missed_limit: u32 = 4; + +/// Priority for the SDIO CLIC line. `hal.intr.init` leaves the threshold at 0 and the comparison +/// is inclusive, so 1 is the lowest value that can ever be taken. Nothing higher would win against +/// anything: `port.zig` is the only owner of a CLIC line in this image. +const sdio_clic_priority: u3 = 1; + +/// Lines each distinct diagnostic may print. A wait that gives up has to be able to say why; it +/// does not have to say so ten thousand times. +const sdio_mark_budget: u32 = 8; + +/// Arming windows between `MARK PORT_SDIO_TALLY` lines. 64 windows is at most 1.3 s of idle link +/// at `sdio_relook_ms`, and far less when frames are flowing, so the ratio is visible within a +/// couple of seconds of boot and costs one `ets_printf` per 64 windows. +const sdio_tally_every: u32 = 64; + +/// Cadence of the polling fallback. Both the pad and the latch are single register reads, so this +/// is a latency budget rather than a cost. +const sdio_poll_ms: u32 = 1; + +/// How long one arming window sleeps before looking at the pad and the latch itself. +/// +/// The number is a latency budget, not a timeout: an interrupt that arrives is delivered at once, +/// and this only bounds how long a *lost* negedge can go unnoticed. 20 ms is two orders of +/// magnitude below anything the transport's own retries care about (`transport_drv.c:233` sleeps +/// 200 ms per turn) and two orders above the cost of the register reads it gates. +const sdio_relook_ms: u32 = 20; + +fn sdioMark(budget: *u32, comptime fmt: [*:0]const u8, args: anytype) void { + if (budget.* >= sdio_mark_budget) return; + budget.* += 1; + note(fmt, args); +} + +/// The interrupt-free path. `sdio_drv.c:1188` insists a finite wait is unusable here, so an +/// unbounded `ticks_to_wait` blocks until the card really does call - it just yields between +/// checks instead of sleeping on a futex. +/// +/// The clear before returning is load-bearing. `sdio_clear_intr` writes the *slave's* +/// `ESP_SLAVE_INT_CLR_REG` (`sdio_drv.c:423-427`); nothing in the C touches this controller's +/// RINTSTS, so a latched bit left set here makes the next call return immediately, and +/// `sdio_read_task`'s loop contains no other yield. That is the same total starvation the +/// interrupt storm caused, reached the slow way - and it is why the interrupt path clears at the +/// top of every arm rather than on the way out. +fn sdioPoll(ticks_to_wait: u32) c_int { + _ = ticks_to_wait; + + // Fast path: if either controller-side signal says the card is calling, say so at once. Both + // are real when they do fire, and they cost two register reads. + if (hal.sdmmc.slaveInterruptAsserted() or hal.sdmmc.slaveInterruptPending()) { + hal.sdmmc.clearSlaveInterrupt(); + return esp_ok; + } + + // Otherwise sleep briefly and report "look again" - deliberately, and this is the whole point of + // this function. + // + // Neither controller-side signal is a trustworthy answer to "does the slave have a packet": + // + // - `slaveInterruptPending` reads RINTSTS bit 16+slot, which LATCHES an edge. The card asserts + // once per packet; clear that latch while the card still has data queued and the edge is + // gone, with nothing to re-create it until the *next* packet arrives. + // - `slaveInterruptAsserted` reads D1's pad level, and D1 is a DATA line. The SDMMC controller + // owns that pad throughout every CMD53, and the SDIO interrupt is only meaningful in defined + // windows between blocks. ESP-IDF never reads it for this: `sdmmc_host_io_int_wait` consults + // the controller's own status word instead. + // + // Measured consequence of trusting them: the receive counter reached somewhere between 6 and 18 + // frames and then froze for ever, while transmits kept working. The board took a real DHCP lease + // - the host speaks first there - and then answered no ARP and no ping. + // + // The authority on "is there a packet" is the slave's own ESP_SLAVE_INT_RAW_REG, and + // `sdio_read_task` already reads it on every pass and tests BIT(SDIO_INT_NEW_PACKET) itself + // (sdio_drv.c:1204, :1247). examples/sdiocheck.zig proved that register answers reliably over + // CMD53. So when the cheap signals say nothing, the right move is not to guess - it is to yield + // and let the caller ask the slave. `HOSTED_BLOCK_MAX` is honoured in the sense that matters: + // this returns only when the caller has something to do, and "read your registers again" always + // is. + // + // The cost is one register read per `sdio_poll_ms` while the link is idle. The benefit is that a + // lost edge can no longer strand a packet. + state.io.sleep(.fromMilliseconds(sdio_poll_ms), os.clock) catch return esp_fail; + return esp_ok; +} + +/// Take ownership of the interrupt controller, once, before any line this file configures can be +/// delivered. +/// +/// **This is the step whose absence made the interrupt path look like an INTMASK write that would +/// not stick.** `hal.intr.init()` is not decoration; it is what makes an interrupt reach *this +/// image* at all, and nothing in the `-Dapp=examples/http.zig` build had ever called it. +/// `examples/intrcheck.zig` and `examples/portcheck.zig` do; `examples/http.zig`, +/// `examples/radio.zig` and everything under `src/` did not. So when +/// `hal.intr.setEnabled(sdio_clic_line, true)` ran on this board, four separate preconditions were +/// missing: +/// +/// * **mtvec still belonged to the bootloader.** `hal.intr.init` fills the vector table, writes +/// MTVT and writes `mtvec = trapEntry | 3` (`intr.zig:558-577`). Without it, the CLIC vectors +/// wherever the ROM left mtvec pointing, `hal.intr.setHandler` files `sdioDispatch` in a table +/// the core never consults, and the core leaves this image and does not come back. That is the +/// reported "whole board going silent right after Open data path at slave": not a storm, an +/// exit. +/// * **whatever the ROM armed was still armed** (`intr.zig:512-518`), so the first MIE could +/// also deliver somebody else's level-triggered source into the same nowhere. +/// * **the threshold was never opened.** The comparison is inclusive and this line runs at +/// priority 1, so a threshold the ROM left at 1 or above masks it for ever - which is a LAPSE +/// every window with no handler entry and nothing else wrong anywhere. +/// * **mstatus.MIE.** `hal.intr.init` deliberately leaves it clear and says that turning it on +/// is the caller's decision (`intr.zig:531`). Nothing in this build was that caller. +/// +/// Enabling MIE here is safe *because* `init()` ran first: it has just detached all 128 sources +/// and cleared all 48 enables, so the only lines that can be delivered afterwards are the ones +/// this file enables itself. +/// +/// Idempotent, and the test is the fact that matters rather than a flag of our own - if mtvec +/// already points at this image's trap entry then somebody has already done this, and re-running +/// `init()` would destroy `hal.intr.boot_state`, the only record of what the bootloader handed +/// over. Both call sites (here and `hostedConfigGpioAsInterrupt`) run it before they touch a line, +/// so whichever is first does the work and the other finds it done - which matters, because +/// `init()` detaches every source and would otherwise silence a line the other had just armed. +fn takeInterruptControl() void { + if (hal.intr.readMtvec() != (hal.intr.trapEntryAddress() | hal.intr.mtvec_mode_clic)) { + hal.intr.init(); + // A fault is the one failure on this path that cannot report itself: `hal.intr` parks the + // core with the numbers recorded and no way to print them. Only installed if the + // application has not claimed the hook. + if (hal.intr.on_fault == null) hal.intr.on_fault = reportFault; + note("MARK PORT_INTR_OWN mtvec=0x%08x want=0x%08x mtvt=0x%08x thresh=%u boot_mie=%u rom_lines=0x%08x rom_sources=%u\r\n", .{ + hal.intr.readMtvec(), + hal.intr.trapEntryAddress() | hal.intr.mtvec_mode_clic, + hal.intr.readMtvt(), + @as(u32, hal.intr.getThreshold()), + @as(u32, @intFromBool(hal.intr.boot_state.mie)), + hal.intr.boot_state.enabled_lines, + hal.intr.boot_state.routed_sources, + }); + } + if (!hal.intr.globalEnabled()) hal.intr.globalEnable(); +} + +/// Last words. `hal.intr.intrFault` has already recorded the fault and will park the core after +/// this returns, so this is the only chance the numbers get to leave the board. +fn reportFault(f: hal.intr.Fault) void { + note("MARK PORT_INTR_FAULT mcause=0x%08x mepc=0x%08x mtval=0x%08x taken=%u last_id=%u spurious=%u\r\n", .{ + f.mcause, + f.mepc, + f.mtval, + hal.intr.taken, + hal.intr.last_clic_id, + hal.intr.spurious, + }); +} + +/// Deassert and disable, in that order. The guarantee the handler needs: after this the line +/// cannot be taken again until somebody arms it. +fn disarmSdioLine() void { + hal.sdmmc.setSlaveInterruptEnabled(false); + hal.intr.setEnabled(config.sdio_clic_line, false); +} + +var sdio_line_configured: bool = false; + +/// The CLIC handler. Runs with `mstatus.MIE` clear on the interrupted stack +/// (`hal.intr.Handler`), so what follows cannot itself be interrupted - and after the first +/// statement it cannot be re-entered either. +fn sdioDispatch(line: u5) void { + _ = line; + // One load, before the disarm, and it is safe for a reason worth stating rather than assuming. + // + // The invariant is "no path returns from this handler with the line still asserted", because a + // level line re-enters the instant the handler `mret`s and that hangs the core. What breaks the + // invariant is a *branch* - any test that can return early. A read cannot return, so a load + // placed here costs the invariant nothing. + // + // It has to be here, though: MINTSTS is `RINTSTS & INTMASK`, so the disarm below zeroes it and + // reading it afterwards would report 0 on every entry - the same mistake the old INTMASK read + // made one line lower. This is the register the controller's output actually follows, so its + // value at the moment of delivery is the direct answer to "did the card interrupt reach the + // CLIC, or did something else". + const mintsts_at_entry = hal.sdmmc.interruptStatusMasked(); + + // Unconditional, and first among the *stores*. A level-triggered line does not deassert because + // the handler returned; masking the source and dropping the CLIC's enable are the only two + // things that stop it, and this handler does not know which status bit is holding the line up. + // Every test placed before this point is a chance to return with the line still asserted, which + // is not a missed interrupt - it is a hang of the whole core. + disarmSdioLine(); + + // The rest of what the line looked like at entry, and the reason + // `hal.sdmmc.interruptStatusRaw` and `hal.sdmmc.dmaStatusRaw` exist. Both of these registers + // are sticky, so reading them after the disarm loses nothing. + // + // INTMASK is deliberately *not* read here. It is not sticky, the disarm has just rewritten it, + // and a `MARK PORT_SDIO_FOREIGN` carrying that value only ever said that the disarm worked. The + // mask that was actually in force is `state.sdio_armed_intmask`, read back by the arm inside its + // own masked region. + state.sdio_intr_mintsts.store(mintsts_at_entry, .release); + state.sdio_intr_rintsts.store(hal.sdmmc.interruptStatusRaw(), .release); + state.sdio_intr_idsts.store(hal.sdmmc.dmaStatusRaw(), .release); + + // Wake unconditionally too. The waiter can tell a real card interrupt from a foreign one, and + // a waiter that is told is a waiter that can report; returning silently is how the old handler + // turned a misconfigured mask into a wait that never ended. + _ = state.sdio_intr_epoch.fetchAdd(1, .release); + state.io.futexWake(u32, &state.sdio_intr_epoch.raw, 1); +} + +// ============================================================================ 7. events + +fn hostedEventWifiPost(event_id: i32, event_data: ?*anyopaque, event_data_size: usize, ticks_to_wait: u32) callconv(.c) c_int { + _ = ticks_to_wait; + deliver(.{ + .base = .wifi, + .id = event_id, + .data = sliceOf(event_data, event_data_size), + }); + return esp_ok; +} + +fn hostedEventPost(event_base: EventBase, event_id: i32, event_data: ?*anyopaque, event_data_size: usize, ticks_to_wait: u32) callconv(.c) c_int { + _ = ticks_to_wait; + deliver(.{ + .base = .{ .named = event_base }, + .id = event_id, + .data = sliceOf(event_data, event_data_size), + }); + return esp_ok; +} + +fn sliceOf(p: ?*anyopaque, len: usize) ?[]const u8 { + const q = p orelse return null; + if (len == 0) return null; + const b: [*]const u8 = @ptrCast(q); + return b[0..len]; +} + +/// `ticks_to_wait` is dropped, and that is a real difference. `esp_event_post` copies the payload +/// into a queue and can block when that queue is full, which is what the argument is for. This +/// calls the application straight through, on the posting task, so there is no queue to fill and +/// nothing to wait for - but it also means a slow handler stalls the transport task that posted the +/// event. The application is expected to copy what it needs and return. +fn deliver(e: Event) void { + const h = state.on_event orelse { + // Silent by default would hide association and disconnection reasons, which is exactly + // what a bring-up needs to see. + switch (e.base) { + .wifi => note("MARK PORT_EVENT wifi id=%d len=%u (no handler)\r\n", .{ e.id, @as(u32, @intCast(if (e.data) |d| d.len else 0)) }), + .named => |n| note("MARK PORT_EVENT %s id=%d len=%u (no handler)\r\n", .{ n, e.id, @as(u32, @intCast(if (e.data) |d| d.len else 0)) }), + } + return; + }; + h(e); +} + +// ============================================================================ misc real entries + +/// `hosted_init_hook` warns if `CONFIG_FREERTOS_HZ` is below ESP-Hosted's recommendation +/// (`port_esp_hosted_host_os.c:150-158`). There is no tick here at all - `std.Io`'s timebase is +/// `hal.systimer`'s 16 MHz counter and sleeps are absolute deadlines, not tick counts - so the +/// jitter that warning is about does not exist. Announce the port instead, which is the one line +/// that proves this table is the one being called. +fn hostedInitHook() callconv(.c) void { + note("MARK PORT_HOOK zig port installed=%u timers=%u\r\n", .{ + @as(u32, @intFromBool(state.installed)), + @as(u32, config.timer_slots), + }); +} + +/// `_h_restart_host` reboots the host when the slave has stopped answering +/// (`transport_drv.c:70`, `sdio_drv.c:578`, and the init-timeout callback). +/// +/// ESP-IDF calls `esp_restart`. There is no `esp_restart` here and, more to the point, a bring-up +/// that silently reboots is a bring-up you cannot debug: the interesting state is the state at the +/// moment the slave went quiet. So this reports and parks, with interrupts left on so the console +/// still works and a debugger can still attach. +fn hostedRestartHost() callconv(.c) c_int { + const s = stats(); + note("MARK PORT_RESTART_HOST requested; parking. heap live=%u reserved=%u peak=%u blocks=%u fail=%u stubs=%u\r\n", .{ + @as(u32, @intCast(s.bytes_live)), + @as(u32, @intCast(s.bytes_reserved)), + @as(u32, @intCast(s.peak_reserved)), + @as(u32, @intCast(s.blocks_live)), + @as(u32, @intCast(s.alloc_failures)), + s.stub_calls, + }); + while (true) {} +} + +/// `_h_get_host_wakeup_or_reboot_reason`. `HOSTED_WAKEUP_NORMAL_REBOOT` is what ESP-IDF returns +/// when power-save is not compiled in (`port_esp_hosted_host_os.c:932-934`), and it is the truth +/// here: this image has no sleep support, so every boot is a normal one. +fn hostedGetWakeupReason() callconv(.c) c_int { + return 0; // HOSTED_WAKEUP_NORMAL_REBOOT +} + +// ============================================================================ 8. loud stubs + +/// Every stub prints its own name and returns a failure code. The two properties that matter: a +/// path nobody implemented is *visible* on the console rather than a hang, and the pointer is never +/// null, so a call through it cannot be a jump to address zero. +fn stub(comptime name: []const u8) void { + state.stub_calls += 1; + note("MARK PORT_STUB " ++ name ++ "\r\n", .{}); +} + +/// SPI only. ESP-IDF assigns this just once, under `H_TRANSPORT_IN_USE == H_TRANSPORT_SPI` +/// (`port_esp_hosted_host_os.c:991`), leaving it **null** for SDIO - so under IDF, reaching this on +/// an SDIO build is a jump to zero. Here it is a message. +fn stubDoBusTransfer(_: ?*anyopaque) callconv(.c) c_int { + stub("_h_do_bus_transfer (SPI transport)"); + return esp_fail; +} + +/// `_h_printf` routes ESP-Hosted's logging through the port table. Nothing in the tree calls it - +/// every `ESP_LOG*` goes to `esp_log_writev` directly, which is the parent's symbol - so this is +/// unreachable in practice, and implementing it would mean either a printf formatter in Zig or a +/// `va_list` handed across an ABI boundary that has not been validated on rv32. The tag and the +/// unexpanded format string are printed, which is enough to identify the call site if it ever +/// happens. +fn stubPrintf(level: c_int, tag: [*:0]const u8, format: [*:0]const u8, ...) callconv(.c) void { + state.stub_calls += 1; + note("MARK PORT_STUB _h_printf level=%d tag=%s fmt=%s (varargs not expanded)\r\n", .{ level, tag, format }); +} + +fn stubSpiHdReadReg(_: u32, _: *u32, _: c_int, _: bool) callconv(.c) c_int { + stub("_h_spi_hd_read_reg"); + return esp_fail; +} +fn stubSpiHdWriteReg(_: u32, _: *u32, _: bool) callconv(.c) c_int { + stub("_h_spi_hd_write_reg"); + return esp_fail; +} +fn stubSpiHdReadDma(_: [*]u8, _: u16, _: bool) callconv(.c) c_int { + stub("_h_spi_hd_read_dma"); + return esp_fail; +} +fn stubSpiHdWriteDma(_: [*]u8, _: u16, _: bool) callconv(.c) c_int { + stub("_h_spi_hd_write_dma"); + return esp_fail; +} +fn stubSpiHdSetDataLines(_: u32) callconv(.c) c_int { + stub("_h_spi_hd_set_data_lines"); + return esp_fail; +} +fn stubSpiHdSendCmd9() callconv(.c) c_int { + stub("_h_spi_hd_send_cmd9"); + return esp_fail; +} + +fn stubUartRead(_: ?*anyopaque, _: [*]u8, _: u16) callconv(.c) c_int { + stub("_h_uart_read"); + return esp_fail; +} +fn stubUartWrite(_: ?*anyopaque, _: [*]u8, _: u16) callconv(.c) c_int { + stub("_h_uart_write"); + return esp_fail; +} +fn stubUartFlushInput(_: ?*anyopaque) callconv(.c) c_int { + stub("_h_uart_flush_input"); + return esp_fail; +} + +/// Power save needs `esp_sleep`, a wakeup GPIO in the LP domain, and a hold latch this HAL does not +/// model. ESP-IDF's own version returns -1 unless `H_HOST_PS_ALLOWED` +/// (`port_esp_hosted_host_os.c:876-891`), so -1 is also the configured-off answer. +fn stubConfigHostPowerSave(_: u32, _: ?*anyopaque, _: u32, _: c_int) callconv(.c) c_int { + stub("_h_config_host_power_save_hal_impl"); + return -1; +} +fn stubStartHostPowerSave(_: u32) callconv(.c) c_int { + stub("_h_start_host_power_save_hal_impl"); + return -1; +} + +// ============================================================================ compile-time census + +/// A compile-time list of which entries are real and which are loud stubs, so the census in the +/// module header cannot drift from the table. `port.stubbed` is what a self-test prints. +pub const stubbed = [_][]const u8{ + "_h_do_bus_transfer", + "_h_printf", + "_h_hold_gpio", + "_h_spi_hd_read_reg", + "_h_spi_hd_write_reg", + "_h_spi_hd_read_dma", + "_h_spi_hd_write_dma", + "_h_spi_hd_set_data_lines", + "_h_spi_hd_send_cmd9", + "_h_uart_read", + "_h_uart_write", + "_h_uart_flush_input", + "_h_config_host_power_save_hal_impl", + "_h_start_host_power_save_hal_impl", +}; + +comptime { + // 71 entries, 14 stubbed, 57 real. + assert(stubbed.len == 14); + assert(std.meta.fields(HostedOsiFuncs).len - stubbed.len == 57); +} diff --git a/src/oracle/all.zig b/src/oracle/all.zig new file mode 100644 index 0000000..9537eac --- /dev/null +++ b/src/oracle/all.zig @@ -0,0 +1,51 @@ +//! Every peripheral registered with the differential harness. +//! +//! One line per peripheral. The harness walks this list, so adding a peripheral to the oracle is +//! three new files (`<name>_ref.c`, `<name>_cases.zig`, `src/hal/<name>.zig`) plus one line here. + +pub const types = @import("differ_types.zig"); + +pub const gpio = @import("gpio_cases.zig"); +pub const clkrst = @import("clkrst_cases.zig"); +pub const timg = @import("timg_cases.zig"); +pub const uart = @import("uart_cases.zig"); +pub const intr = @import("intr_cases.zig"); +pub const ledc = @import("ledc_cases.zig"); +pub const i2c = @import("i2c_cases.zig"); +pub const sdmmc = @import("sdmmc_cases.zig"); + +/// The suites, in the order they run. +/// +/// GPIO first: the console's own pins live in that block, so a failure there explains failures +/// everywhere else. `clkrst` last of the always-on set, because its cases deliberately gate +/// peripherals off and its restore is what puts them back. +pub const suites = [_]types.Suite{ + gpio.suite, + gpio.iomux_suite, + timg.suite, + uart.suite, + intr.suite, + intr.clic_suite, + intr.thresh_suite, + // LEDC's state is not contiguous, so it comes as four windows rather than one: the block + // itself, the gamma RAM aperture (whose restore has to zero the RAM, because a peripheral reset + // does not), the GPIO words its pin routing touches, and the one HP_SYS_CLKRST word the P4 + // moved its clock mux into. + ledc.suite, + ledc.gamma_suite, + ledc.routing_suite, + ledc.clock_suite, + // I2C likewise needs two: half of setBusTiming lands outside the I2C block, because the + // controller-clock divider is in HP_SYS_CLKRST. At 10 kHz that divider is 4, so an + // implementation that wrote all ten timing registers perfectly and the divider not at all would + // run the bus four times too fast and pass every case in the first suite. + i2c.suite, + i2c.clock_suite, + // SDMMC, likewise in two windows: the controller block, and the host clock generator that the + // P4 put in HP_SYS_CLKRST rather than in the peripheral. At 40 MHz the whole division happens + // in the second one, so a suite that covered only the first would pass on a bus running four + // times too fast. + sdmmc.suite, + sdmmc.clock_suite, + clkrst.suite, +}; diff --git a/src/oracle/clkrst_cases.zig b/src/oracle/clkrst_cases.zig new file mode 100644 index 0000000..f004dd4 --- /dev/null +++ b/src/oracle/clkrst_cases.zig @@ -0,0 +1,131 @@ +//! HP_SYS_CLKRST's side of the differential: the clock-gate and reset pairing table. +//! +//! This suite exists because of a bug that a hardware test failed to catch. `src/hal/clkrst.zig` +//! maps each peripheral to the register and bit that gate and reset it, and every row of that table +//! is a transcription from ESP-IDF's LL source - the field macros do not record which register they +//! live in, so there is nothing to derive it from. Two rows were wrong: timg0, timg1, systimer and +//! twai0 had their APB clock enables in `PERI_CLK_CTRL21` instead of `SOC_CLK_CTRL2`, so +//! `setClockEnabled` wrote a bit of an unrelated register. +//! +//! `examples/halcheck.zig` printed exactly the expected `twai0 boot=0 on=1 off=0` throughout, +//! because `isClockEnabled` read back the same wrong bit `setClockEnabled` had just written. A +//! self-consistent test proves the two halves of your own code agree; it does not prove either one +//! touches the hardware, and that one would have passed with the chip unplugged. +//! +//! ESP-IDF reaches these bits through its own generated struct definitions - a genuinely independent +//! path - so this comparison is the check a read-back cannot be. +//! +//! Not covered here: TWAI0. Its bus clock is the one that is gated off at power-on, which makes it +//! the interesting case, but ESP-IDF's TWAI bus-clock LL takes a controller handle this suite has no +//! business constructing. The four rows below share the two registers TWAI0's row uses, so a +//! transcription error in it would have to be independent of theirs to survive. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +extern fn oracle_clkrst_timg_bus_clock(group: c_uint, enable: c_int) void; +extern fn oracle_clkrst_timg_reset(group: c_uint) void; +extern fn oracle_clkrst_systimer_bus_clock(enable: c_int) void; +extern fn oracle_clkrst_systimer_reset() void; +extern fn oracle_clkrst_uart_bus_clock(port: c_uint, enable: c_int) void; + +/// Known state: every peripheral this suite touches with its bus clock on, which is also the state +/// the chip powers up in ("All peripheral clocks are default enabled after chip is powered on", +/// esp_system/port/soc/esp32p4/clk.c:200). Nothing else in the block is touched - UART0's gates in +/// particular, because that is the console this result is printed over. +/// Restore through ESP-IDF's LL, never through the code under test. +/// +/// This suite exists to catch a `setClockEnabled` that writes the wrong register. Restoring with +/// `hal.clkrst.setClockEnabled` defeated exactly that: `differ.zig` runs restore, idf, snapshot, +/// restore, ours, snapshot, so with the HAL on both the restore and the "ours" side, a +/// `setClockEnabled` that did nothing at all would leave run B's snapshot equal to run A's and pass +/// all six clock cases. Which is how the original bug - four peripherals' gate bits in +/// PERI_CLK_CTRL21 instead of SOC_CLK_CTRL2 - could have survived this suite too. +fn restore() void { + oracle_clkrst_timg_bus_clock(1, 1); + oracle_clkrst_systimer_bus_clock(1); + oracle_clkrst_uart_bus_clock(1, 1); +} + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "clkrst", + .base = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG - 0x18), // block base + // 0x00 through HP_RST_EN2 at +0xC8: covers SOC_CLK_CTRL1/2 (+0x18, +0x1c), every + // PERI_CLK_CTRL register, and all three HP_RST_EN registers. Everything either + // implementation could plausibly hit is inside this window, which is the property that + // makes a difference detectable rather than merely absent. + .words = 52, + .restore = .{ .configure = restore }, + }, + .cases = &.{ + // Disable first in each pair: the restored state has them on, so "disable" is the operation + // with an observable effect and "enable" would otherwise be a no-op comparison. + .{ .name = "timg1_bus_clock", .arg = 0, .idf = idfTimgOff, .ours = ourTimgOff }, + .{ .name = "systimer_bus_clock", .arg = 0, .idf = idfSystimerOff, .ours = ourSystimerOff }, + .{ .name = "uart1_bus_clock", .arg = 0, .idf = idfUartOff, .ours = ourUartOff }, + .{ .name = "timg1_reset", .idf = idfTimgReset, .ours = ourTimgReset }, + .{ .name = "systimer_reset", .idf = idfSystimerReset, .ours = ourSystimerReset }, + // Last, so the block is left with everything on regardless of which side ran last. + .{ .name = "timg1_bus_clock", .arg = 1, .idf = idfTimgOn, .ours = ourTimgOn }, + .{ .name = "systimer_bus_clock", .arg = 1, .idf = idfSystimerOn, .ours = ourSystimerOn }, + .{ .name = "uart1_bus_clock", .arg = 1, .idf = idfUartOn, .ours = ourUartOn }, + }, +}; + +fn idfTimgOff() void { + oracle_clkrst_timg_bus_clock(1, 0); +} +fn ourTimgOff() void { + hal.clkrst.setClockEnabled(.timg1, false); +} +fn idfTimgOn() void { + oracle_clkrst_timg_bus_clock(1, 1); +} +fn ourTimgOn() void { + hal.clkrst.setClockEnabled(.timg1, true); +} +fn idfSystimerOff() void { + oracle_clkrst_systimer_bus_clock(0); +} +fn ourSystimerOff() void { + hal.clkrst.setClockEnabled(.systimer, false); +} +fn idfSystimerOn() void { + oracle_clkrst_systimer_bus_clock(1); +} +fn ourSystimerOn() void { + hal.clkrst.setClockEnabled(.systimer, true); +} +fn idfUartOff() void { + oracle_clkrst_uart_bus_clock(1, 0); +} +fn ourUartOff() void { + hal.clkrst.setClockEnabled(.uart1, false); +} +fn idfUartOn() void { + oracle_clkrst_uart_bus_clock(1, 1); +} +fn ourUartOn() void { + hal.clkrst.setClockEnabled(.uart1, true); +} + +/// The reset pairing, which is the other half of the table and the half that was right. IDF pulses +/// the bit and returns; so does ours, except for the timer groups, where it additionally clears the +/// flash-boot watchdog protection that the reset re-arms - so a difference in the WDT register is +/// expected and lives outside this window, while HP_RST_EN1 itself must match. +fn idfTimgReset() void { + oracle_clkrst_timg_reset(1); +} +fn ourTimgReset() void { + hal.clkrst.resetPeripheral(.timg1); +} +fn idfSystimerReset() void { + oracle_clkrst_systimer_reset(); +} +fn ourSystimerReset() void { + hal.clkrst.resetPeripheral(.systimer); +} diff --git a/src/oracle/clkrst_ref.c b/src/oracle/clkrst_ref.c new file mode 100644 index 0000000..b35436c --- /dev/null +++ b/src/oracle/clkrst_ref.c @@ -0,0 +1,55 @@ +/* ESP-IDF's own bus-clock and reset control, as the reference for src/hal/clkrst.zig. + * + * This suite exists because of a specific bug. `clkrst.zig` maps each peripheral to the register and + * bit that gate and reset it, and that table is hand-written: the field macros do not say which + * register they live in, so every row is a transcription from ESP-IDF's LL source. Two rows were + * wrong. The second batch - timg0, timg1, systimer and twai0 - had their APB clock enables in + * HP_SYS_CLKRST_PERI_CLK_CTRL21 instead of SOC_CLK_CTRL2, so `setClockEnabled` poked a bit of an + * unrelated register. + * + * It survived a hardware test, which is the point. `examples/halcheck.zig` printed exactly the + * expected `twai0 boot=0 on=1 off=0`, because the write and the read-back both went through the same + * wrong address: a self-consistent test that would have passed with the chip unplugged. + * + * IDF reaches these bits by a completely independent path - its own generated struct definitions - + * so comparing against it is the check that a read-back cannot be. + */ + +/* IDF shadows every clock/reset LL function with a macro that references this identifier, which it + * deliberately never defines, so that an unguarded call fails to compile: the only legal caller + * holds a spinlock. There is no FreeRTOS here and core 1 is held in reset at power-on, so declaring + * the name is exactly as safe as the lock would be. IDF's own bootloader does the same thing + * (bootloader_support/src/bootloader_console.c:53). */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); +/* IDF uses a second name for the same trick on the peripherals whose gate lives in a register shared + * with the CPU's own clocking - systimer among them. Same reasoning applies. */ +static int __DECLARE_RCC_RC_ATOMIC_ENV __attribute__((unused)); + +#include "hal/timg_ll.h" +#include "hal/systimer_ll.h" +#include "hal/uart_ll.h" + +void oracle_clkrst_timg_bus_clock(unsigned group, int enable) +{ + _timg_ll_enable_bus_clock(group, enable != 0); +} + +void oracle_clkrst_timg_reset(unsigned group) +{ + _timg_ll_reset_register(group); +} + +void oracle_clkrst_systimer_bus_clock(int enable) +{ + systimer_ll_enable_bus_clock(enable != 0); +} + +void oracle_clkrst_systimer_reset(void) +{ + systimer_ll_reset_register(); +} + +void oracle_clkrst_uart_bus_clock(unsigned port, int enable) +{ + _uart_ll_enable_bus_clock(port, enable != 0); +} diff --git a/src/oracle/differ_types.zig b/src/oracle/differ_types.zig new file mode 100644 index 0000000..6f8e03c --- /dev/null +++ b/src/oracle/differ_types.zig @@ -0,0 +1,72 @@ +//! The contract between the differential harness and a peripheral under test. +//! +//! Adding a peripheral to the oracle is three files and no edits to the harness: +//! +//! src/oracle/<name>_ref.c external-linkage wrappers over ESP-IDF's `*_ll.h` functions +//! src/oracle/<name>_cases.zig a `descriptor` and a `cases` array, both of the types below +//! src/hal/<name>.zig this project's implementation, which is what is being tested +//! +//! The harness then, for every case: brings the peripheral to a known state, runs ESP-IDF's version, +//! photographs the register block, restores, runs ours, photographs again, and compares. + +/// Everything the harness needs to test a peripheral without breaking the board. +pub const Peripheral = struct { + name: [*:0]const u8, + + /// First address of the register block, and how many 32-bit words to compare. + base: u32, + words: u32, + + /// Word offsets that must never be *read*, because reading them changes hardware state. + /// + /// This cannot be derived from the headers: `UART_FIFO_REG` sits at offset 0 of every UART + /// block, its only field is annotated `RO`, and reading it pops the RX FIFO. A generic + /// block-snapshot loop over a UART eats received bytes - including on the console. + no_read: []const u32 = &.{}, + + /// Word offsets whose value legitimately changes between two runs: counters, FIFO depths, live + /// input levels. Compared they would produce noise, so they are excluded. + volatile_words: []const u32 = &.{}, + + /// The bus-clock enable bit that must read 1 for a snapshot to mean anything. + /// + /// Reading a clock-gated block does not fault and does not return zeros - it returns the last + /// value latched, so two snapshots of a gated peripheral can compare *equal* while describing + /// nothing. The harness checks this before every comparison and fails the case if it is clear. + clock: ?Bit = null, + + /// How to return the peripheral to a known state between the two implementations. + restore: Restore, + + pub const Bit = struct { reg: u32, bit: u5 }; + + pub const Restore = union(enum) { + /// Pulse the peripheral's reset bit in HP_SYS_CLKRST. The only sound restore for a block + /// with write-to-trigger or write-only fields, because it is what the datasheet defines the + /// reset values against. Writing a snapshot back is *not* an option: ~10% of this chip's + /// fields perform an action when written, and writing one saved word back to a UART's + /// offset 0 transmits a character. + reset_bit: Bit, + /// A function that configures the block to a fixed state. For peripherals with no reset bit + /// of their own (GPIO, IO_MUX) or where resetting would take the console with it (UART0). + configure: *const fn () void, + }; +}; + +/// One operation, expressed twice: ESP-IDF's way and ours. They must be the same operation with the +/// same arguments, or the comparison means nothing. +pub const Case = struct { + name: [*:0]const u8, + /// Printed with the result, so a failure names the arguments that produced it. + arg: u32 = 0, + idf: *const fn () void, + ours: *const fn () void, +}; + +/// What a `<name>_cases.zig` module must expose. +pub const Suite = struct { + descriptor: Peripheral, + cases: []const Case, + /// Run once before the suite: bring the peripheral far enough up that its registers are live. + setup: ?*const fn () void = null, +}; diff --git a/src/oracle/gpio_cases.zig b/src/oracle/gpio_cases.zig new file mode 100644 index 0000000..ea32eff --- /dev/null +++ b/src/oracle/gpio_cases.zig @@ -0,0 +1,233 @@ +//! GPIO's side of the differential test: the same operations expressed as ESP-IDF's LL calls and as +//! this project's HAL calls. +//! +//! GPIO is restored by configuring rather than by resetting. It has no reset bit of its own in +//! HP_SYS_CLKRST, and the pads are the board's wiring - the console's own pins are in this block, so +//! a reset here would take the console with it. Configuring is sound for GPIO specifically because +//! every field in the block is plain read/write: there is nothing self-clearing to restore. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +extern fn oracle_gpio_uses_rom_api() c_int; +extern fn oracle_gpio_set_level(pin: c_uint, level: c_uint) void; +extern fn oracle_gpio_output_enable(pin: c_uint) void; +extern fn oracle_gpio_output_disable(pin: c_uint) void; +extern fn oracle_gpio_input_enable(pin: c_uint) void; +extern fn oracle_gpio_input_disable(pin: c_uint) void; +extern fn oracle_gpio_func_sel(pin: c_uint, func: c_uint) void; +extern fn oracle_gpio_set_drive(pin: c_uint, strength: c_uint) void; +extern fn oracle_gpio_pullup_en(pin: c_uint) void; +extern fn oracle_gpio_pullup_dis(pin: c_uint) void; +extern fn oracle_gpio_pulldown_en(pin: c_uint) void; +extern fn oracle_gpio_pulldown_dis(pin: c_uint) void; +extern fn oracle_gpio_matrix_out(pin: c_uint, signal: c_uint) void; +extern fn oracle_gpio_od_enable(pin: c_uint) void; +extern fn oracle_gpio_od_disable(pin: c_uint) void; + +/// Whether ESP-IDF's LL was compiled to call the mask ROM instead of writing registers. Must be 0, +/// or the differential is comparing this HAL against `rom_gpio_set_output_level` rather than against +/// IDF's register sequence. Governed by src/oracle/oracle_sdkconfig.h. +pub fn usesRomApi() bool { + return oracle_gpio_uses_rom_api() != 0; +} + +/// The pin under test. A module-level variable because Zig has no closures and the harness stores +/// plain `fn` pointers: a comptime-specialised pair per pin would compare code this project does not +/// ship instead of the code it does. +pub var pin: u8 = 20; + +/// Pins worth testing. 20 is the board's LED pin and 33 is a free header pin above the 32-boundary +/// where this peripheral's bank arithmetic changes. GPIO54 is deliberately absent: it is this +/// board's ESP32-C6 reset line, held high by an external pull-up, and driving it resets the radio. +pub const pins = [_]u8{ 20, 33 }; + +/// Restore, built from register macros only. +/// +/// Nothing here may call the code under test. `differ.zig` runs restore, idf, snapshot, restore, +/// ours, snapshot - so if restore is written with the HAL, run B starts from whatever IDF just wrote +/// and a HAL function that does nothing at all compares equal. This suite used to restore with +/// `hal.gpio.outputDisable` and `hal.gpio.setLow`, which made `output_disable` and `set_level(0)` +/// no-op-versus-no-op: they could not fail. +/// +/// It must also be *total* over everything any case touches. Leaving `GPIO_PIN{n}_REG` alone made +/// both `open_drain` cases vacuous, because run B inherited run A's pad_driver bit. +fn restore() void { + const b: u5 = @intCast(if (pin < 32) pin else pin - 32); + const m = @as(u32, 1) << b; + const enable_w1tc = if (pin < 32) regs.GPIO_ENABLE_W1TC_REG else regs.GPIO_ENABLE1_W1TC_REG; + const out_w1tc = if (pin < 32) regs.GPIO_OUT_W1TC_REG else regs.GPIO_OUT1_W1TC_REG; + mmio.Reg.atAddress(@intCast(enable_w1tc)).writeRaw(m); + mmio.Reg.atAddress(@intCast(out_w1tc)).writeRaw(m); + // The IO MUX pad word, the matrix output selector, and the GPIO block's own per-pin register. + mmio.Reg.atAddress(@as(u32, @intCast(regs.PERIPHS_IO_MUX_U_PAD_GPIO0)) + 4 * @as(u32, pin)).writeRaw(0); + mmio.Reg.atAddress(@as(u32, @intCast(regs.GPIO_FUNC0_OUT_SEL_CFG_REG)) + 4 * @as(u32, pin)) + .writeRaw(@intCast(regs.SIG_GPIO_OUT_IDX)); + mmio.Reg.atAddress(@as(u32, @intCast(regs.GPIO_PIN0_REG)) + 4 * @as(u32, pin)).writeRaw(0); +} + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "gpio", + .base = @intCast(regs.GPIO_OUT_REG - 4), // GPIO_BT_SELECT_REG sits at +0x00 + // 0x640 bytes. The window has to reach 0x558 + 4*57, where the matrix's per-pad output + // configuration lives: a first version stopped at 0x1C0 and was blind to a real bug in + // exactly those words - it saw the redundant GPIO_ENABLE write but not the wrong OEN_SEL + // that made it necessary. + .words = 400, + .volatile_words = &.{ + (0x03c - 0x000) / 4, // GPIO_IN - reflects the outside world, which moves + (0x040 - 0x000) / 4, // GPIO_IN1 + }, + .restore = .{ .configure = restore }, + }, + .cases = &.{ + .{ .name = "set_level", .arg = 1, .idf = idfSetHigh, .ours = ourSetHigh }, + .{ .name = "set_level", .arg = 0, .idf = idfSetLow, .ours = ourSetLow }, + .{ .name = "output_enable", .idf = idfOutEnable, .ours = ourOutEnable }, + .{ .name = "output_disable", .idf = idfOutDisable, .ours = ourOutDisable }, + .{ .name = "input_enable", .idf = idfInEnable, .ours = ourInEnable }, + .{ .name = "input_disable", .idf = idfInDisable, .ours = ourInDisable }, + .{ .name = "func_sel_gpio", .arg = 1, .idf = idfFuncGpio, .ours = ourFuncGpio }, + .{ .name = "drive", .arg = 3, .idf = idfDriveStrong, .ours = ourDriveStrong }, + .{ .name = "drive", .arg = 0, .idf = idfDriveWeakest, .ours = ourDriveWeakest }, + .{ .name = "pull_up", .idf = idfPullUp, .ours = ourPullUp }, + .{ .name = "pull_down", .idf = idfPullDown, .ours = ourPullDown }, + .{ .name = "pull_none", .idf = idfPullNone, .ours = ourPullNone }, + .{ .name = "matrix_out", .arg = 43, .idf = idfMatrixOut, .ours = ourMatrixOut }, + // Open drain lives in the GPIO block's per-pin register, not the IO MUX pad register, and + // had no accessor until the I2C port needed one - that bus is wired-AND, and a pin left + // push-pull shorts it against another device's driver. + .{ .name = "open_drain", .arg = 1, .idf = idfOdOn, .ours = ourOdOn }, + .{ .name = "open_drain", .arg = 0, .idf = idfOdOff, .ours = ourOdOff }, + }, +}; + +/// The IO MUX, which the GPIO block's window does not reach. +/// +/// Every pad-configuration function on this chip writes `IO_MUX.gpio[n]` at +/// PERIPHS_IO_MUX_U_PAD_GPIO0 = 0x500E1004 + 4*pin, and the GPIO block's compared window ends at +/// 0x500E063F - 0xC00 bytes short. So `input_enable`, `input_disable`, `func_sel`, both `drive` +/// cases and all three `pull` cases were comparing two identical snapshots of a register file none +/// of them touches: 8 operations across 2 pins, 16 of the suite's cases, structurally unable to +/// fail. They are the same cases; only the window is different. +pub const iomux_suite: types.Suite = .{ + .descriptor = .{ + .name = "iomux", + .base = @intCast(regs.PERIPHS_IO_MUX_U_PAD_GPIO0), + .words = 57, // one per pad, GPIO0..GPIO56 + .restore = .{ .configure = restoreIomux }, + }, + .cases = &.{ + .{ .name = "input_enable", .idf = idfInEnable, .ours = ourInEnable }, + .{ .name = "input_disable", .idf = idfInDisable, .ours = ourInDisable }, + .{ .name = "func_sel_gpio", .arg = 1, .idf = idfFuncGpio, .ours = ourFuncGpio }, + .{ .name = "drive", .arg = 3, .idf = idfDriveStrong, .ours = ourDriveStrong }, + .{ .name = "drive", .arg = 0, .idf = idfDriveWeakest, .ours = ourDriveWeakest }, + .{ .name = "pull_up", .idf = idfPullUp, .ours = ourPullUp }, + .{ .name = "pull_down", .idf = idfPullDown, .ours = ourPullDown }, + .{ .name = "pull_none", .idf = idfPullNone, .ours = ourPullNone }, + }, +}; + +fn restoreIomux() void { + mmio.Reg.atAddress(@as(u32, @intCast(regs.PERIPHS_IO_MUX_U_PAD_GPIO0)) + 4 * @as(u32, pin)).writeRaw(0); +} + +fn idfSetHigh() void { + oracle_gpio_set_level(pin, 1); +} +fn ourSetHigh() void { + hal.gpio.setHigh(pin); +} +fn idfSetLow() void { + oracle_gpio_set_level(pin, 0); +} +fn ourSetLow() void { + hal.gpio.setLow(pin); +} +fn idfOutEnable() void { + oracle_gpio_output_enable(pin); +} +fn ourOutEnable() void { + hal.gpio.outputEnable(pin); +} +fn idfOutDisable() void { + oracle_gpio_output_disable(pin); +} +fn ourOutDisable() void { + hal.gpio.outputDisable(pin); +} +fn idfInEnable() void { + oracle_gpio_input_enable(pin); +} +fn ourInEnable() void { + hal.gpio.setInputEnable(pin, true); +} +fn idfInDisable() void { + oracle_gpio_input_disable(pin); +} +fn ourInDisable() void { + hal.gpio.setInputEnable(pin, false); +} +fn idfFuncGpio() void { + oracle_gpio_func_sel(pin, 1); +} +fn ourFuncGpio() void { + hal.gpio.setFunction(pin, .gpio); +} +fn idfDriveStrong() void { + oracle_gpio_set_drive(pin, 3); +} +fn ourDriveStrong() void { + hal.gpio.setDrive(pin, .strong); +} +fn idfDriveWeakest() void { + oracle_gpio_set_drive(pin, 0); +} +fn ourDriveWeakest() void { + hal.gpio.setDrive(pin, .weakest); +} +fn idfPullUp() void { + oracle_gpio_pullup_en(pin); + oracle_gpio_pulldown_dis(pin); +} +fn ourPullUp() void { + hal.gpio.setPull(pin, .up); +} +fn idfPullDown() void { + oracle_gpio_pulldown_en(pin); + oracle_gpio_pullup_dis(pin); +} +fn ourPullDown() void { + hal.gpio.setPull(pin, .down); +} +fn idfPullNone() void { + oracle_gpio_pullup_dis(pin); + oracle_gpio_pulldown_dis(pin); +} +fn ourPullNone() void { + hal.gpio.setPull(pin, .none); +} +fn idfMatrixOut() void { + oracle_gpio_matrix_out(pin, 43); +} +fn ourMatrixOut() void { + hal.gpio.matrixOut(pin, 43); +} + +fn idfOdOn() void { + oracle_gpio_od_enable(pin); +} +fn ourOdOn() void { + hal.gpio.setOpenDrain(pin, true); +} +fn idfOdOff() void { + oracle_gpio_od_disable(pin); +} +fn ourOdOff() void { + hal.gpio.setOpenDrain(pin, false); +} diff --git a/src/oracle/gpio_ref.c b/src/oracle/gpio_ref.c new file mode 100644 index 0000000..b4070a2 --- /dev/null +++ b/src/oracle/gpio_ref.c @@ -0,0 +1,116 @@ +/* The reference implementation, which is ESP-IDF's own. + * + * ESP-IDF's `*_ll.h` headers are `static inline` functions over the same registers this project's + * Zig HAL drives. Compiled by Zig's clang for riscv32-freestanding they link into the same image as + * the Zig code, which is what makes a differential test possible at all: one binary, one boot, one + * set of clocks, both implementations, and the diff taken on the die. + * + * These wrappers exist only to give the inline functions external linkage so Zig can call them. + * There is no logic here - anything clever in this file would be a third implementation to doubt. + */ + +/* IDF's clock and reset LL functions are shadowed by a wrapper macro that references + * `__DECLARE_RCC_ATOMIC_ENV`, an identifier IDF never defines anywhere; its purpose is to make an + * unguarded call fail to compile, because the only legal caller holds a spinlock. There is no + * FreeRTOS here, and core 1 is held in reset at power-on, so declaring the name is exactly as safe + * as the spinlock would be - and it is what IDF's own bootloader does + * (bootloader_support/src/bootloader_console.c:53 declares a dummy local for the same reason). */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); + +#include "hal/gpio_ll.h" +#include "soc/gpio_struct.h" +#include "soc/io_mux_struct.h" + +/* Whether this translation unit was built with the ROM path switched on. The harness prints it, so + * that a differential run can never silently be "my registers versus the mask ROM". */ +int oracle_gpio_uses_rom_api(void) +{ +#if HAL_CONFIG(GPIO_USE_ROM_API) + return 1; +#else + return 0; +#endif +} + +void oracle_gpio_set_level(unsigned pin, unsigned level) +{ + gpio_ll_set_level(&GPIO, pin, level); +} + +int oracle_gpio_get_level(unsigned pin) +{ + return gpio_ll_get_level(&GPIO, pin); +} + +void oracle_gpio_output_enable(unsigned pin) +{ + gpio_ll_output_enable(&GPIO, pin); +} + +void oracle_gpio_output_disable(unsigned pin) +{ + gpio_ll_output_disable(&GPIO, pin); +} + +void oracle_gpio_input_enable(unsigned pin) +{ + gpio_ll_input_enable(&GPIO, pin); +} + +void oracle_gpio_input_disable(unsigned pin) +{ + gpio_ll_input_disable(&GPIO, pin); +} + +void oracle_gpio_func_sel(unsigned pin, unsigned func) +{ + gpio_ll_func_sel(&GPIO, pin, func); +} + +void oracle_gpio_set_drive(unsigned pin, unsigned strength) +{ + gpio_ll_set_drive_capability(&GPIO, pin, (gpio_drive_cap_t)strength); +} + +void oracle_gpio_pullup_en(unsigned pin) +{ + gpio_ll_pullup_en(&GPIO, pin); +} + +void oracle_gpio_pullup_dis(unsigned pin) +{ + gpio_ll_pullup_dis(&GPIO, pin); +} + +void oracle_gpio_pulldown_en(unsigned pin) +{ + gpio_ll_pulldown_en(&GPIO, pin); +} + +void oracle_gpio_pulldown_dis(unsigned pin) +{ + gpio_ll_pulldown_dis(&GPIO, pin); +} + +/* Open drain, which lives in the GPIO block's own per-pin register (GPIO_PINn_PAD_DRIVER) rather + * than in the IO MUX pad register - a different register file for the same pad. The I2C HAL needs it + * because that bus is wired-AND, and a pin left push-pull shorts a shared bus against another + * device's driver. One bit, and expensive to get wrong. */ +void oracle_gpio_od_enable(unsigned pin) +{ + gpio_ll_od_enable(&GPIO, pin); +} + +void oracle_gpio_od_disable(unsigned pin) +{ + gpio_ll_od_disable(&GPIO, pin); +} + +/* Route a peripheral signal to a pad through the GPIO matrix. This is the one GPIO operation with a + * real sequence rather than a single field write, and therefore the one where a write-trace + * comparison can find something a state comparison cannot. */ +void oracle_gpio_matrix_out(unsigned pin, unsigned signal) +{ + gpio_ll_set_output_signal_matrix_source(&GPIO, pin, signal, false); + gpio_ll_set_output_enable_ctrl(&GPIO, pin, true, false); +} diff --git a/src/oracle/i2c_cases.zig b/src/oracle/i2c_cases.zig new file mode 100644 index 0000000..0546066 --- /dev/null +++ b/src/oracle/i2c_cases.zig @@ -0,0 +1,627 @@ +//! I2C's side of the differential test: the same operations expressed as ESP-IDF's LL calls and as +//! this project's HAL calls. +//! +//! Two suites, because this peripheral's state lives in two register blocks that are 0x24000 bytes +//! apart and the harness compares one window per suite: +//! +//! * `suite` - the I2C0 block itself (0x500C4000, 128 words). Timing, FIFOs, the command list, +//! the filter, the timeout. +//! * `clock_suite` - the two HP_SYS_CLKRST words that hold I2C's controller clock: source select, +//! clock enable and the divider, for *both* ports (HP_SYS_CLKRST_PERI_CLK_CTRL10/11). Without +//! this second window the divider half of `setBusTiming` would be untested, because the divider +//! write does not land in the I2C block at all. Registering only the first suite would leave a +//! bus that is a factor of `clkm_div` too fast with nothing to notice. +//! +//! Restore differs between the two, and both choices are forced: +//! +//! * The I2C block is restored by its **reset bit**. It has three write-to-trigger fields +//! (`trans_start`, `fsm_rst`, `conf_upgate`) and a self-setting `command_done` per slot, so +//! writing a snapshot back would trigger a transaction. HP_SYS_CLKRST's reset bit is what the +//! datasheet defines the reset values against, and I2C0 carries nothing this board needs - no +//! console, no flash - so pulsing it is safe. +//! * The clock words cannot be reset that way: they are in HP_SYS_CLKRST, not in the I2C block, and +//! PERI_CLK_CTRL11 also holds three I2S0_RX clock fields. Restore there is a configure function +//! that writes only I2C's own fields back to their documented reset value of zero. +//! +//! The restore function deliberately builds its field descriptors from the macros itself rather than +//! calling into `hal.i2c`: a restore that shared the HAL's idea of where a field lives would agree +//! with a HAL that had it wrong, and the case would pass while configuring the wrong bits. Same +//! reason `i2c_ref.c` maps command *kinds* to IDF's `I2C_LL_CMD_*` macros instead of taking an +//! opcode number from Zig. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +extern fn oracle_i2c_enable_bus_clock(port: c_int, enable: c_int) void; +extern fn oracle_i2c_reset_register(port: c_int) void; +extern fn oracle_i2c_enable_controller_clock(port: c_int, enable: c_int) void; +extern fn oracle_i2c_set_source_clk(port: c_int, src: c_int) void; +extern fn oracle_i2c_master_init(port: c_int) void; +extern fn oracle_i2c_set_mode_master(port: c_int) void; +extern fn oracle_i2c_enable_pins_open_drain(port: c_int, enable_od: c_int) void; +extern fn oracle_i2c_update(port: c_int) void; +extern fn oracle_i2c_fsm_rst(port: c_int) void; +extern fn oracle_i2c_set_bus_timing(port: c_int, source_hz: c_uint, bus_hz: c_uint) void; +extern fn oracle_i2c_set_start_timing(port: c_int, setup: c_int, hold: c_int) void; +extern fn oracle_i2c_set_stop_timing(port: c_int, setup: c_int, hold: c_int) void; +extern fn oracle_i2c_set_sda_timing(port: c_int, sample: c_int, hold: c_int) void; +extern fn oracle_i2c_set_tout(port: c_int, tout: c_int) void; +extern fn oracle_i2c_set_scl_timeout_us(port: c_int, source_hz: c_uint, timeout_us: c_uint) void; +extern fn oracle_i2c_set_filter(port: c_int, filter_num: c_uint) void; +extern fn oracle_i2c_txfifo_rst(port: c_int) void; +extern fn oracle_i2c_rxfifo_rst(port: c_int) void; +extern fn oracle_i2c_enable_fifo_mode(port: c_int, fifo_mode_en: c_int) void; +extern fn oracle_i2c_set_fifo_thresholds(port: c_int, tx_empty: c_uint, rx_full: c_uint) void; +extern fn oracle_i2c_write_txfifo_pattern(port: c_int, len: c_uint) void; +extern fn oracle_i2c_write_cmd( + port: c_int, + slot: c_int, + kind: c_uint, + byte_num: c_uint, + ack_en: c_int, + ack_exp: c_int, + ack_val: c_int, +) void; +extern fn oracle_i2c_clear_intr_mask(port: c_int, mask: c_uint) void; +extern fn oracle_i2c_disable_intr_mask(port: c_int, mask: c_uint) void; +extern fn oracle_i2c_get_hw_version(port: c_int) c_uint; +extern fn oracle_i2c_cmd_reg_num() c_uint; +extern fn oracle_i2c_fifo_len() c_uint; + +/// ESP-IDF's own view of two chip constants this HAL hard-codes. The harness prints them; a +/// disagreement means `hal.i2c.cmd_slots` or `fifo_len` was read out of the wrong chip's header, +/// which is a mistake no register comparison would ever show. +pub fn idfCmdSlots() u32 { + return oracle_i2c_cmd_reg_num(); +} + +pub fn idfFifoLen() u32 { + return oracle_i2c_fifo_len(); +} + +pub fn hardwareVersion() u32 { + return oracle_i2c_get_hw_version(0); +} + +comptime { + // These are constants in both implementations, so they can be checked here rather than on the + // die - but only against the *header*, which is why the runtime accessors above exist too. + if (hal.i2c.cmd_slots != 8) @compileError("this chip has eight command slots"); + if (hal.i2c.fifo_len != 32) @compileError("this chip's I2C FIFO is 32 bytes"); +} + +// There is no module-level "port under test" variable here, unlike the GPIO suite's `pin`, and the +// reason is in the descriptors: a `Peripheral` carries one `base` and one `clock`, both constants, +// so the I2C-block suite is pinned to I2C0 by construction and running it "for port 1" would need a +// second descriptor rather than a variable. Nothing is lost by that, because the only per-port +// arithmetic in this peripheral is which HP_SYS_CLKRST field a port's clock lives in - and both +// ports' fields are inside `clock_suite`'s two-word window, where the cases name the port directly. + +/// 40 MHz crystal, which is what `Timing.calculate` is fed on both sides. Not a measurement: the +/// board's crystal, and the P4's only XTAL frequency. +const source_hz: u32 = hal.i2c.xtal_hz; + +// --------------------------------------------------------------------------- the I2C0 block + +fn resetI2c0() void { + // Same pulse the harness's `.reset_bit` restore performs, for `setup` to use before the first + // case. Interrupt-masked because HP_RST_EN1 holds every peripheral's reset bit. + const guard = hal.clkrst.maskInterrupts(); + defer guard.release(); + const r = Reg.at(regs.HP_SYS_CLKRST_HP_RST_EN1_REG); + const bit = @as(u32, 1) << @intCast(regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_S); + r.writeRaw(r.raw() | bit); + r.writeRaw(r.raw() & ~bit); +} + +/// Bring I2C0 far enough up that its registers are live and its state machine is clocked. +/// +/// The APB gate defaults to 1 on this chip so the registers are readable from boot, but the +/// *controller* clock defaults to 0 - and that one is in HP_SYS_CLKRST, outside the block, so the +/// reset-bit restore between cases does not disturb it. +fn setupI2c0() void { + hal.clkrst.setClockEnabled(.i2c0, true); + hal.i2c.setControllerClockEnabled(0, true); + resetI2c0(); +} + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "i2c", + .base = @intCast(regs.I2C_SCL_LOW_PERIOD_REG(0)), // I2C0 + 0x000 + // 128 words = 0x200 bytes, which is the whole instance: configuration and the command list + // end at +0x84, the version word is at +0xf8, and the two 32-byte FIFO RAMs are at +0x100 + // (TX) and +0x180 (RX). The RAMs are in the window on purpose - a TX FIFO write is otherwise + // observable only as a count in I2C_SR, and a count is a much weaker witness than the bytes + // themselves. If those words ever turn out to read unstably in FIFO mode - ESP-IDF only ever + // touches them in non-FIFO mode - they belong in `volatile_words`, not out of the window. + .words = 128, + // Reading I2C_DATA_REG pops the RX FIFO. The register header gives no hint of it: the only + // field is annotated `HRO` and described as "Rx FIFO read data" (i2c_reg.h:464-474). What + // settles it is that `i2c_ll_read_rxfifo` reads this one address `len` times and expects + // `len` different bytes (i2c_ll.h:691-697), which is only possible if the read advances the + // FIFO - and `i2c_ll_write_txfifo` writes the same address to fill the *other* FIFO + // (i2c_ll.h:674-680). Same shape as UART_FIFO_REG. A snapshot loop that reads it would eat + // received bytes and desynchronise the read pointer under the case being measured. + .no_read = &.{hal.i2c.data_word_offset}, + .clock = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL2_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_S), + }, + .restore = .{ .reset_bit = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_HP_RST_EN1_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_S), + } }, + }, + .cases = &.{ + // ---- bus timing. Five frequencies, chosen for the branches rather than for roundness. + // 100 kHz and 400 kHz are the two speeds every device supports; 1 MHz is fast-mode-plus, + // where half_cycle is down to 20 source cycles and the minus-one asymmetries dominate; + // 50 kHz and 10 kHz are on the other side of the 80 kHz boundary where the scl_wait_high + // split changes formula (i2c_ll.h:112-115); and 10 kHz is the one that needs a controller + // clock divider greater than 1 - the half that this window cannot see, which is what + // `clock_suite` is for. + .{ .name = "bus_timing_100k", .arg = 100_000, .idf = idfTiming100k, .ours = ourTiming100k }, + .{ .name = "bus_timing_400k", .arg = 400_000, .idf = idfTiming400k, .ours = ourTiming400k }, + .{ .name = "bus_timing_1M", .arg = 1_000_000, .idf = idfTiming1M, .ours = ourTiming1M }, + .{ .name = "bus_timing_50k", .arg = 50_000, .idf = idfTiming50k, .ours = ourTiming50k }, + .{ .name = "bus_timing_10k", .arg = 10_000, .idf = idfTiming10k, .ours = ourTiming10k }, + + // ---- master bring-up, and the open-drain polarity on its own. + .{ .name = "master_init", .idf = idfMasterInit, .ours = ourMasterInit }, + .{ .name = "pins_open_drain", .arg = 1, .idf = idfOpenDrainOn, .ours = ourOpenDrainOn }, + .{ .name = "pins_push_pull", .arg = 0, .idf = idfOpenDrainOff, .ours = ourOpenDrainOff }, + .{ .name = "fifo_mode", .arg = 1, .idf = idfFifoMode, .ours = ourFifoMode }, + .{ .name = "nonfifo_mode", .arg = 0, .idf = idfNonFifoMode, .ours = ourNonFifoMode }, + + // ---- FIFOs. The resets are two stores each (the bit is not self-clearing), so a + // half-done reset shows up as a FIFO held in reset rather than as a wrong value. + .{ .name = "txfifo_rst", .idf = idfTxFifoRst, .ours = ourTxFifoRst }, + .{ .name = "rxfifo_rst", .idf = idfRxFifoRst, .ours = ourRxFifoRst }, + .{ .name = "txfifo_write", .arg = 4, .idf = idfWrite4, .ours = ourWrite4 }, + .{ .name = "txfifo_write", .arg = 31, .idf = idfWrite31, .ours = ourWrite31 }, + .{ .name = "fifo_thresholds", .arg = 8, .idf = idfThresholds, .ours = ourThresholds }, + + // ---- filter. Three cases because "off" is not "on with a threshold of zero": both enables + // default to 1 with zero thresholds, so disabling has to clear the enables and leave the + // thresholds alone (i2c_ll.h:753-764). + .{ .name = "filter_7", .arg = 7, .idf = idfFilter7, .ours = ourFilter7 }, + .{ .name = "filter_15", .arg = 15, .idf = idfFilter15, .ours = ourFilter15 }, + .{ .name = "filter_off", .arg = 0, .idf = idfFilter0, .ours = ourFilter0 }, + + // ---- timeout. The field is five bits and holds an *exponent*: the bus times out after + // 2^value source-clock cycles, so 12 is 102 us at 40 MHz and 31 is the largest the register + // can hold. The third case goes through the microsecond conversion IDF's driver uses + // (i2c_ll.h:1060-1065) for its documented 2000 us default, which comes out as 17. + .{ .name = "tout_12", .arg = 12, .idf = idfTout12, .ours = ourTout12 }, + .{ .name = "tout_31", .arg = 31, .idf = idfTout31, .ours = ourTout31 }, + .{ .name = "scl_timeout_us", .arg = 2000, .idf = idfSclTimeoutUs, .ours = ourSclTimeoutUs }, + + // ---- the explicit timing setters, where IDF's minus-one convention is least uniform: + // start setup as given but start hold minus one, stop and sda both as given. + .{ .name = "start_timing", .arg = 7, .idf = idfStartTiming, .ours = ourStartTiming }, + .{ .name = "stop_timing", .arg = 5, .idf = idfStopTiming, .ours = ourStopTiming }, + .{ .name = "sda_timing", .arg = 11, .idf = idfSdaTiming, .ours = ourSdaTiming }, + + // ---- the command list, one opcode per slot. The IDF side names the opcode + // (`I2C_LL_CMD_*`) and the ours side names it too (`Op.restart`), so the *numbers* are never + // passed across: this chip's register header documents the pre-C3 numbering, and a test that + // handed the number over would agree with a wrong constant instead of catching it. + .{ .name = "cmd_restart", .arg = 0, .idf = idfCmdRestart, .ours = ourCmdRestart }, + .{ .name = "cmd_write_ack", .arg = 5, .idf = idfCmdWrite, .ours = ourCmdWrite }, + .{ .name = "cmd_read_ack", .arg = 3, .idf = idfCmdReadAck, .ours = ourCmdReadAck }, + .{ .name = "cmd_read_nack", .arg = 1, .idf = idfCmdReadNack, .ours = ourCmdReadNack }, + .{ .name = "cmd_stop", .arg = 0, .idf = idfCmdStop, .ours = ourCmdStop }, + .{ .name = "cmd_end", .arg = 0, .idf = idfCmdEnd, .ours = ourCmdEnd }, + .{ .name = "cmd_list_write", .arg = 4, .idf = idfCmdListWrite, .ours = ourCmdListWrite }, + + // ---- interrupt state. Not an interrupt-driven driver - this HAL polls - but the clear + // register is write-1-to-clear, so getting it wrong (a read-modify-write instead of a raw + // store) is a class of bug worth one case. + .{ .name = "clear_intr", .idf = idfClearIntr, .ours = ourClearIntr }, + .{ .name = "disable_intr", .idf = idfDisableIntr, .ours = ourDisableIntr }, + }, + .setup = setupI2c0, +}; + +// ---- bus timing -------------------------------------------------------------------------------- +// Each pair is IDF's calculate-and-write (i2c_hal.c:27-32) against ours (hal.i2c.setBusTiming). The +// comparison covers ten in-block registers at once, so a single wrong subtraction anywhere in the +// derivation shows up here. + +fn idfTiming100k() void { + oracle_i2c_set_bus_timing(0, source_hz, 100_000); +} +fn ourTiming100k() void { + hal.i2c.setBusTiming(0, source_hz, 100_000); +} +fn idfTiming400k() void { + oracle_i2c_set_bus_timing(0, source_hz, 400_000); +} +fn ourTiming400k() void { + hal.i2c.setBusTiming(0, source_hz, 400_000); +} +fn idfTiming1M() void { + oracle_i2c_set_bus_timing(0, source_hz, 1_000_000); +} +fn ourTiming1M() void { + hal.i2c.setBusTiming(0, source_hz, 1_000_000); +} +fn idfTiming50k() void { + oracle_i2c_set_bus_timing(0, source_hz, 50_000); +} +fn ourTiming50k() void { + hal.i2c.setBusTiming(0, source_hz, 50_000); +} +fn idfTiming10k() void { + oracle_i2c_set_bus_timing(0, source_hz, 10_000); +} +fn ourTiming10k() void { + hal.i2c.setBusTiming(0, source_hz, 10_000); +} + +// ---- bring-up ---------------------------------------------------------------------------------- + +fn idfMasterInit() void { + oracle_i2c_master_init(0); +} +fn ourMasterInit() void { + hal.i2c.initMaster(0); +} +fn idfOpenDrainOn() void { + oracle_i2c_enable_pins_open_drain(0, 1); +} +fn ourOpenDrainOn() void { + hal.i2c.setPinsOpenDrain(0, true); +} +fn idfOpenDrainOff() void { + oracle_i2c_enable_pins_open_drain(0, 0); +} +fn ourOpenDrainOff() void { + hal.i2c.setPinsOpenDrain(0, false); +} +fn idfFifoMode() void { + oracle_i2c_enable_fifo_mode(0, 1); +} +fn ourFifoMode() void { + hal.i2c.setFifoMode(0, true); +} +fn idfNonFifoMode() void { + oracle_i2c_enable_fifo_mode(0, 0); +} +fn ourNonFifoMode() void { + hal.i2c.setFifoMode(0, false); +} + +// ---- FIFOs ------------------------------------------------------------------------------------- + +fn idfTxFifoRst() void { + oracle_i2c_txfifo_rst(0); +} +fn ourTxFifoRst() void { + hal.i2c.resetTxFifo(0); +} +fn idfRxFifoRst() void { + oracle_i2c_rxfifo_rst(0); +} +fn ourRxFifoRst() void { + hal.i2c.resetRxFifo(0); +} + +/// The same pattern `oracle_i2c_write_txfifo_pattern` generates: 0xA0 + i, so every byte differs +/// from its neighbours and from the 0x00/0xFF a broken FIFO produces. +const pattern: [hal.i2c.fifo_len]u8 = blk: { + var p: [hal.i2c.fifo_len]u8 = undefined; + for (&p, 0..) |*b, i| b.* = 0xA0 + @as(u8, @intCast(i)); + break :blk p; +}; + +fn idfWrite4() void { + oracle_i2c_write_txfifo_pattern(0, 4); +} +fn ourWrite4() void { + hal.i2c.writeTxFifo(0, pattern[0..4]); +} +// 31 bytes rather than 32: one short of full, so the case cannot be passed by a FIFO that silently +// wrapped and cannot trip the overflow protection either. +fn idfWrite31() void { + oracle_i2c_write_txfifo_pattern(0, 31); +} +fn ourWrite31() void { + hal.i2c.writeTxFifo(0, pattern[0..31]); +} +fn idfThresholds() void { + oracle_i2c_set_fifo_thresholds(0, 8, 20); +} +fn ourThresholds() void { + hal.i2c.setFifoThresholds(0, 8, 20); +} + +// ---- filter and timeout ------------------------------------------------------------------------ + +fn idfFilter7() void { + oracle_i2c_set_filter(0, 7); +} +fn ourFilter7() void { + hal.i2c.setFilter(0, 7); +} +fn idfFilter15() void { + oracle_i2c_set_filter(0, 15); +} +fn ourFilter15() void { + hal.i2c.setFilter(0, 15); +} +fn idfFilter0() void { + oracle_i2c_set_filter(0, 0); +} +fn ourFilter0() void { + hal.i2c.setFilter(0, 0); +} +fn idfTout12() void { + oracle_i2c_set_tout(0, 12); +} +fn ourTout12() void { + hal.i2c.setTimeout(0, 12); +} +fn idfTout31() void { + oracle_i2c_set_tout(0, 31); +} +fn ourTout31() void { + hal.i2c.setTimeout(0, 31); +} +fn idfSclTimeoutUs() void { + oracle_i2c_set_scl_timeout_us(0, source_hz, 2000); +} +fn ourSclTimeoutUs() void { + hal.i2c.setTimeout(0, hal.i2c.timeoutExponent(source_hz, 2000)); +} + +// ---- explicit timing setters ------------------------------------------------------------------- +// The same registers `applyTiming` writes, but reached by IDF's three narrow setters, whose +// minus-one convention is *different* from the one in the calculate-and-write path: start setup as +// given and start hold minus one, both stop values as given, both sda values as given +// (i2c_ll.h:452-486 against i2c_ll.h:210-217). Numbers with no relation to any real bus frequency, +// so a HAL that quietly recomputed them from a frequency instead of writing what it was given would +// show up here rather than passing. + +fn idfStartTiming() void { + oracle_i2c_set_start_timing(0, 7, 9); +} +fn ourStartTiming() void { + hal.i2c.setStartTiming(0, 7, 9); +} +fn idfStopTiming() void { + oracle_i2c_set_stop_timing(0, 5, 6); +} +fn ourStopTiming() void { + hal.i2c.setStopTiming(0, 5, 6); +} +fn idfSdaTiming() void { + oracle_i2c_set_sda_timing(0, 11, 3); +} +fn ourSdaTiming() void { + hal.i2c.setSdaTiming(0, 11, 3); +} + +// ---- the command list -------------------------------------------------------------------------- + +// Kind numbers as `i2c_ref.c` reads them: 0 restart, 1 write, 2 read, 3 stop, 4 end. Only the *kind* +// crosses the language boundary; the C side turns it into an opcode with IDF's own macro. +const kind_restart: c_uint = 0; +const kind_write: c_uint = 1; +const kind_read: c_uint = 2; +const kind_stop: c_uint = 3; +const kind_end: c_uint = 4; + +fn idfCmdRestart() void { + oracle_i2c_write_cmd(0, 0, kind_restart, 0, 0, 0, 0); +} +fn ourCmdRestart() void { + hal.i2c.writeCommand(0, 0, .{ .op = .restart }); +} +fn idfCmdWrite() void { + oracle_i2c_write_cmd(0, 1, kind_write, 5, 1, 0, 0); +} +fn ourCmdWrite() void { + hal.i2c.writeCommand(0, 1, .{ .op = .write, .bytes = 5, .ack_check = true }); +} +fn idfCmdReadAck() void { + oracle_i2c_write_cmd(0, 2, kind_read, 3, 0, 0, 0); +} +fn ourCmdReadAck() void { + hal.i2c.writeCommand(0, 2, .{ .op = .read, .bytes = 3, .ack_value = 0 }); +} +fn idfCmdReadNack() void { + oracle_i2c_write_cmd(0, 3, kind_read, 1, 0, 0, 1); +} +fn ourCmdReadNack() void { + hal.i2c.writeCommand(0, 3, .{ .op = .read, .bytes = 1, .ack_value = 1 }); +} +fn idfCmdStop() void { + oracle_i2c_write_cmd(0, 4, kind_stop, 0, 0, 0, 0); +} +fn ourCmdStop() void { + hal.i2c.writeCommand(0, 4, .{ .op = .stop }); +} +fn idfCmdEnd() void { + oracle_i2c_write_cmd(0, 5, kind_end, 0, 0, 0, 0); +} +fn ourCmdEnd() void { + hal.i2c.writeCommand(0, 5, .{ .op = .end }); +} + +/// A whole list, in the shape `hal.i2c.write` builds for a four-byte transfer: RSTART, WRITE of +/// 1 + 4 bytes with ACK checking, STOP. Slots 3 to 7 keep the reset value on both sides. +fn idfCmdListWrite() void { + oracle_i2c_write_cmd(0, 0, kind_restart, 0, 0, 0, 0); + oracle_i2c_write_cmd(0, 1, kind_write, 5, 1, 0, 0); + oracle_i2c_write_cmd(0, 2, kind_stop, 0, 0, 0, 0); +} +fn ourCmdListWrite() void { + hal.i2c.writeCommands(0, &.{ + .{ .op = .restart }, + .{ .op = .write, .bytes = 5, .ack_check = true }, + .{ .op = .stop }, + }); +} + +// ---- interrupt state --------------------------------------------------------------------------- + +fn idfClearIntr() void { + oracle_i2c_clear_intr_mask(0, hal.i2c.all_interrupts); +} +fn ourClearIntr() void { + hal.i2c.clearInterrupts(0, hal.i2c.all_interrupts); +} +fn idfDisableIntr() void { + oracle_i2c_disable_intr_mask(0, hal.i2c.all_interrupts); +} +fn ourDisableIntr() void { + hal.i2c.disableInterrupts(0); +} + +// ------------------------------------------------------- the clock domain: HP_SYS_CLKRST words +// +// Field descriptors built here rather than borrowed from hal.i2c, on purpose: the restore function +// below must not share the HAL's idea of where these fields live, or a HAL with a field in the wrong +// place would be restored consistently with its own mistake and every case would pass. + +const peri_clk_ctrl10 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL10_REG); +const peri_clk_ctrl11 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL11_REG); + +const i2c0_clock_fields = [_]Field{ + Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_V), +}; + +const i2c1_divider_fields = [_]Field{ + Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_V), + Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_V), +}; + +/// Zero every I2C clock field in the two words, which is their documented reset value +/// (hp_sys_clkrst_reg.h: all ten default to 0), leaving everything else in those words alone. +/// +/// "Everything else" is not empty: PERI_CLK_CTRL11 also holds `REG_I2S0_RX_CLK_EN` and +/// `REG_I2S0_RX_CLK_SRC_SEL` in bits 24-26. Restoring by writing a whole word would take I2S0's +/// receive clock with it, which is exactly the class of collateral damage the harness's +/// no-write-back rule exists to prevent - so this is a masked read-modify-write, interrupt-masked +/// because these registers are shared. +fn restoreI2cClocks() void { + const guard = hal.clkrst.maskInterrupts(); + defer guard.release(); + var mask10: u32 = 0; + for (i2c0_clock_fields) |f| mask10 |= f.mask(); + peri_clk_ctrl10.writeRaw(peri_clk_ctrl10.raw() & ~mask10); + var mask11: u32 = 0; + for (i2c1_divider_fields) |f| mask11 |= f.mask(); + peri_clk_ctrl11.writeRaw(peri_clk_ctrl11.raw() & ~mask11); +} + +/// I2C's controller clock: source, gate and divider, for both ports, in two words. +/// +/// This is the other half of `setBusTiming`. The divider is what keeps `half_cycle` inside the +/// nine-bit period fields at low bus frequencies - 10 kHz needs `clkm_div` 4 - so an implementation +/// that wrote the timing registers correctly and the divider not at all would produce a bus four +/// times too fast and pass every case in the suite above. +pub const clock_suite: types.Suite = .{ + .descriptor = .{ + .name = "i2c_clk", + .base = @intCast(regs.HP_SYS_CLKRST_PERI_CLK_CTRL10_REG), + // Two words: CTRL10 (all of I2C0's clock fields plus I2C1's source select and gate) and + // CTRL11 (I2C1's divider, and three I2S0_RX bits neither side touches). + .words = 2, + // No reset bit for HP_SYS_CLKRST, and no gate in front of it either: it is the block that + // holds every other block's gate. + .restore = .{ .configure = restoreI2cClocks }, + }, + .cases = &.{ + .{ .name = "source_xtal", .arg = 0, .idf = idfSourceXtal0, .ours = ourSourceXtal0 }, + .{ .name = "source_rc_fast", .arg = 0, .idf = idfSourceRcFast0, .ours = ourSourceRcFast0 }, + .{ .name = "source_xtal_p1", .arg = 1, .idf = idfSourceXtal1, .ours = ourSourceXtal1 }, + .{ .name = "source_rc_fast_p1", .arg = 1, .idf = idfSourceRcFast1, .ours = ourSourceRcFast1 }, + .{ .name = "controller_clock_on", .arg = 0, .idf = idfCtrlClkOn0, .ours = ourCtrlClkOn0 }, + .{ .name = "controller_clock_off", .arg = 0, .idf = idfCtrlClkOff0, .ours = ourCtrlClkOff0 }, + .{ .name = "controller_clock_on_p1", .arg = 1, .idf = idfCtrlClkOn1, .ours = ourCtrlClkOn1 }, + // The divider written by the same calculate-and-write pair as the timing cases, at the two + // frequencies either side of where clkm_div stops being 1. + .{ .name = "divider_100k", .arg = 100_000, .idf = idfDiv100k, .ours = ourDiv100k }, + .{ .name = "divider_10k", .arg = 10_000, .idf = idfDiv10k, .ours = ourDiv10k }, + // ... and on port 1, where the divider is in the *other* word from its own source select. + .{ .name = "divider_10k_p1", .arg = 10_000, .idf = idfDiv10kP1, .ours = ourDiv10kP1 }, + }, + .setup = null, +}; + +fn idfSourceXtal0() void { + oracle_i2c_set_source_clk(0, 0); +} +fn ourSourceXtal0() void { + hal.i2c.setSource(0, .xtal); +} +fn idfSourceRcFast0() void { + oracle_i2c_set_source_clk(0, 1); +} +fn ourSourceRcFast0() void { + hal.i2c.setSource(0, .rc_fast); +} +fn idfSourceXtal1() void { + oracle_i2c_set_source_clk(1, 0); +} +fn ourSourceXtal1() void { + hal.i2c.setSource(1, .xtal); +} +fn idfSourceRcFast1() void { + oracle_i2c_set_source_clk(1, 1); +} +fn ourSourceRcFast1() void { + hal.i2c.setSource(1, .rc_fast); +} +fn idfCtrlClkOn0() void { + oracle_i2c_enable_controller_clock(0, 1); +} +fn ourCtrlClkOn0() void { + hal.i2c.setControllerClockEnabled(0, true); +} +fn idfCtrlClkOff0() void { + oracle_i2c_enable_controller_clock(0, 0); +} +fn ourCtrlClkOff0() void { + hal.i2c.setControllerClockEnabled(0, false); +} +fn idfCtrlClkOn1() void { + oracle_i2c_enable_controller_clock(1, 1); +} +fn ourCtrlClkOn1() void { + hal.i2c.setControllerClockEnabled(1, true); +} +fn idfDiv100k() void { + oracle_i2c_set_bus_timing(0, source_hz, 100_000); +} +fn ourDiv100k() void { + hal.i2c.setBusTiming(0, source_hz, 100_000); +} +fn idfDiv10k() void { + oracle_i2c_set_bus_timing(0, source_hz, 10_000); +} +fn ourDiv10k() void { + hal.i2c.setBusTiming(0, source_hz, 10_000); +} +fn idfDiv10kP1() void { + oracle_i2c_set_bus_timing(1, source_hz, 10_000); +} +fn ourDiv10kP1() void { + hal.i2c.setBusTiming(1, source_hz, 10_000); +} diff --git a/src/oracle/i2c_ref.c b/src/oracle/i2c_ref.c new file mode 100644 index 0000000..b569260 --- /dev/null +++ b/src/oracle/i2c_ref.c @@ -0,0 +1,262 @@ +/* I2C's reference implementation, which is ESP-IDF's own. + * + * These wrappers exist only to give IDF's `static inline` LL functions external linkage so Zig can + * call them. There is no logic here - anything clever in this file would be a third implementation + * to doubt - with one deliberate exception, `opcode_of`, explained where it appears. + */ + +/* IDF's clock and reset LL functions are shadowed by a wrapper macro that references + * `__DECLARE_RCC_ATOMIC_ENV`, an identifier IDF never defines anywhere; its purpose is to make an + * unguarded call fail to compile, because the only legal caller holds a spinlock. There is no + * FreeRTOS here and core 1 is held in reset at power-on, so declaring the name is exactly as safe as + * the spinlock would be - and it is what IDF's own bootloader does + * (bootloader_support/src/bootloader_console.c:53 declares a dummy local for the same reason). + * + * For I2C this covers four functions: i2c_ll_enable_bus_clock, i2c_ll_reset_register, + * i2c_ll_set_source_clk and the LP_I2C ones this file does not use. */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); + +#include "hal/i2c_ll.h" + +/* I2C0 and I2C1 are addresses PROVIDEd by soc/esp32p4/ld/esp32p4.peripherals.ld (lines 17-18: + * 0x500C4000 and 0x500C5000), which the build links when -Doracle is passed. Several LL functions + * dispatch on the *pointer* - i2c_ll_master_set_bus_timing compares `hw == &I2C0` to decide which + * HP_SYS_CLKRST divider to write - so passing the right one of these two is load-bearing, and a + * third port cannot be faked. */ +static i2c_dev_t *dev(int port) +{ + return (port == 0) ? &I2C0 : &I2C1; +} + +/* ------------------------------------------------------------------ clocks, reset, bring-up */ + +void oracle_i2c_enable_bus_clock(int port, int enable) +{ + i2c_ll_enable_bus_clock(port, enable != 0); +} + +void oracle_i2c_reset_register(int port) +{ + i2c_ll_reset_register(port); +} + +void oracle_i2c_enable_controller_clock(int port, int enable) +{ + i2c_ll_enable_controller_clock(dev(port), enable != 0); +} + +/* src: 0 = XTAL, 1 = RC_FAST. Passed as IDF's own enum values rather than as the register bit, so a + * wrong bit polarity in the Zig would show up as a difference. */ +void oracle_i2c_set_source_clk(int port, int src) +{ + i2c_ll_set_source_clk(dev(port), (src == 1) ? I2C_CLK_SRC_RC_FAST : I2C_CLK_SRC_XTAL); +} + +/* i2c_hal_master_init, i2c_hal.c:39-50, inlined here because i2c_hal.c is not linked into this + * image - only the LL headers are. The sequence is IDF's, unchanged. */ +void oracle_i2c_master_init(int port) +{ + i2c_dev_t *hw = dev(port); + i2c_ll_set_mode(hw, I2C_BUS_MODE_MASTER); + i2c_ll_enable_pins_open_drain(hw, true); + i2c_ll_enable_arbitration(hw, false); + i2c_ll_master_rx_full_ack_level(hw, false); + i2c_ll_set_data_mode(hw, I2C_DATA_MODE_MSB_FIRST, I2C_DATA_MODE_MSB_FIRST); + i2c_ll_txfifo_rst(hw); + i2c_ll_rxfifo_rst(hw); +} + +void oracle_i2c_set_mode_master(int port) +{ + i2c_ll_set_mode(dev(port), I2C_BUS_MODE_MASTER); +} + +void oracle_i2c_enable_pins_open_drain(int port, int enable_od) +{ + i2c_ll_enable_pins_open_drain(dev(port), enable_od != 0); +} + +void oracle_i2c_update(int port) +{ + i2c_ll_update(dev(port)); +} + +void oracle_i2c_fsm_rst(int port) +{ + i2c_ll_master_fsm_rst(dev(port)); +} + +/* ------------------------------------------------------------------------------- bus timing */ + +/* _i2c_hal_set_bus_timing, i2c_hal.c:27-32: calculate then write. The calculation + * (i2c_ll_master_cal_bus_clk, i2c_ll.h:104-128) is the part this project reimplements in Zig, and + * this is the only honest way to compare it - the computed struct never leaves the register file, so + * the comparison has to be of the registers it produced. */ +void oracle_i2c_set_bus_timing(int port, unsigned source_hz, unsigned bus_hz) +{ + i2c_hal_clk_config_t clk_cal = {0}; + i2c_ll_master_cal_bus_clk(source_hz, bus_hz, &clk_cal); + i2c_ll_master_set_bus_timing(dev(port), &clk_cal); +} + +/* The three timing setters that take explicit periods, which is where IDF's minus-one convention is + * least uniform: start_setup is written as given while start_hold is written minus one + * (i2c_ll.h:452-456), both stop values are written as given (i2c_ll.h:467-471), and so are both sda + * values (i2c_ll.h:482-486). None of that is derivable from the register headers. */ +void oracle_i2c_set_start_timing(int port, int setup, int hold) +{ + i2c_ll_master_set_start_timing(dev(port), setup, hold); +} + +void oracle_i2c_set_stop_timing(int port, int setup, int hold) +{ + i2c_ll_master_set_stop_timing(dev(port), setup, hold); +} + +void oracle_i2c_set_sda_timing(int port, int sample, int hold) +{ + i2c_ll_set_sda_timing(dev(port), sample, hold); +} + +void oracle_i2c_set_tout(int port, int tout) +{ + i2c_ll_set_tout(dev(port), tout); +} + +/* i2c_hal_master_set_scl_timeout_val, i2c_hal.c:66-70. */ +void oracle_i2c_set_scl_timeout_us(int port, unsigned source_hz, unsigned timeout_us) +{ + uint32_t reg_val = i2c_ll_calculate_timeout_us_to_reg_val(source_hz, timeout_us); + i2c_ll_set_tout(dev(port), reg_val); +} + +void oracle_i2c_set_filter(int port, unsigned filter_num) +{ + i2c_ll_master_set_filter(dev(port), (uint8_t)filter_num); +} + +/* --------------------------------------------------------------------------------- the FIFOs */ + +void oracle_i2c_txfifo_rst(int port) +{ + i2c_ll_txfifo_rst(dev(port)); +} + +void oracle_i2c_rxfifo_rst(int port) +{ + i2c_ll_rxfifo_rst(dev(port)); +} + +void oracle_i2c_enable_fifo_mode(int port, int fifo_mode_en) +{ + i2c_ll_enable_fifo_mode(dev(port), fifo_mode_en != 0); +} + +void oracle_i2c_set_fifo_thresholds(int port, unsigned tx_empty, unsigned rx_full) +{ + i2c_ll_set_txfifo_empty_thr(dev(port), (uint8_t)tx_empty); + i2c_ll_set_rxfifo_full_thr(dev(port), (uint8_t)rx_full); +} + +/* A pattern rather than a caller-supplied buffer: the point is that both implementations push the + * same bytes through the same FIFO port, and a fixed generator makes the two sides impossible to + * accidentally disagree about. 0xA0 + i is chosen so every byte differs from its neighbours and from + * 0x00/0xFF, which are the values a broken FIFO produces. */ +void oracle_i2c_write_txfifo_pattern(int port, unsigned len) +{ + uint8_t buf[32]; + if (len > sizeof(buf)) { + len = sizeof(buf); + } + for (unsigned i = 0; i < len; i++) { + buf[i] = (uint8_t)(0xA0 + i); + } + i2c_ll_write_txfifo(dev(port), buf, (uint8_t)len); +} + +/* ------------------------------------------------------------------------- the command list */ + +/* The one piece of logic in this file, and it is here on purpose: it maps a command *kind* to + * ESP-IDF's own `I2C_LL_CMD_*` macro, so the opcode number crosses the boundary as a name rather + * than as an integer. If the Zig side had the numbers wrong - and this chip's register header + * documents the pre-ESP32-C3 numbering, so getting them wrong is easy - passing the raw number + * through would make both sides agree on the same mistake and the differential would prove nothing. + * + * Kinds: 0 restart, 1 write, 2 read, 3 stop, 4 end. */ +static uint32_t opcode_of(unsigned kind) +{ + switch (kind) { + case 0: return I2C_LL_CMD_RESTART; + case 1: return I2C_LL_CMD_WRITE; + case 2: return I2C_LL_CMD_READ; + case 3: return I2C_LL_CMD_STOP; + default: return I2C_LL_CMD_END; + } +} + +void oracle_i2c_write_cmd(int port, int slot, unsigned kind, unsigned byte_num, + int ack_en, int ack_exp, int ack_val) +{ + i2c_ll_hw_cmd_t cmd = { + .byte_num = byte_num, + .ack_en = (ack_en != 0), + .ack_exp = (ack_exp != 0), + .ack_val = (ack_val != 0), + .op_code = opcode_of(kind), + }; + i2c_ll_master_write_cmd_reg(dev(port), cmd, slot); +} + +/* ---------------------------------------------------------------------------- interrupt state */ + +void oracle_i2c_clear_intr_mask(int port, unsigned mask) +{ + i2c_ll_clear_intr_mask(dev(port), mask); +} + +void oracle_i2c_disable_intr_mask(int port, unsigned mask) +{ + i2c_ll_disable_intr_mask(dev(port), mask); +} + +/* ------------------------------------------------------------------------------ observations */ + +/* Not part of any comparison - these exist so the harness can print what the reference thinks the + * hardware says, next to what ours says, when a case fails. */ +unsigned oracle_i2c_get_hw_version(int port) +{ + return i2c_ll_get_hw_version(dev(port)); +} + +unsigned oracle_i2c_get_txfifo_len(int port) +{ + uint32_t len = 0; + i2c_ll_get_txfifo_len(dev(port), &len); + return len; +} + +unsigned oracle_i2c_get_rxfifo_cnt(int port) +{ + uint32_t len = 0; + i2c_ll_get_rxfifo_cnt(dev(port), &len); + return len; +} + +int oracle_i2c_get_tout(int port) +{ + int tout = 0; + i2c_ll_get_tout(dev(port), &tout); + return tout; +} + +/* The chip's command-slot count as ESP-IDF's own header states it, so the Zig constant is checked + * against IDF rather than against a reading of IDF. */ +unsigned oracle_i2c_cmd_reg_num(void) +{ + return I2C_LL_CMD_REG_NUM; +} + +unsigned oracle_i2c_fifo_len(void) +{ + return I2C_LL_FIFO_LEN; +} diff --git a/src/oracle/intr_cases.zig b/src/oracle/intr_cases.zig new file mode 100644 index 0000000..78f2965 --- /dev/null +++ b/src/oracle/intr_cases.zig @@ -0,0 +1,595 @@ +//! The interrupt controller's side of the differential test. +//! +//! **What this can and cannot prove.** A register comparison is weaker evidence for an interrupt +//! controller than for any other peripheral here, and pretending otherwise would be the worst thing +//! this file could do. What it establishes is that for each operation below, this HAL leaves the +//! same words behind that ESP-IDF's code does - the matrix address arithmetic, the `+ 16` offset, +//! the two different priority encodings, the trigger encoding, and which bits each operation is +//! allowed to disturb. What it cannot establish is that an interrupt is ever *taken*: that depends +//! on mtvec, MTVT, mstatus.MIE, the trap entry's register save and the CLIC's arbitration, none of +//! which appear in any register this harness photographs. The behavioural test is spelled out at +//! the foot of this file and the parent must run it separately. +//! +//! **Three windows, three suites.** The controller is three disjoint pieces of address space: +//! * the interrupt matrix at DR_REG_INTERRUPT_CORE0_BASE, 128 words, one per source; +//! * the CLIC's global registers at 0x2080_0000, three words, the third being the threshold; +//! * the CLIC's per-interrupt control file at 0x2080_1000, 48 words, one per CLIC ID. +//! A single window spanning them would have to read about a thousand words of address space nothing +//! is mapped at. Each descriptor below covers only registers these cases actually touch. +//! +//! **Nothing here enables interrupts.** No case calls `hal.intr.init()`, sets mstatus.MIE or writes +//! mtvec; the cases enable *lines* at the CLIC, which with MIE clear is inert. `setup` clears MIE +//! explicitly, so that is true even if something earlier in the image set it. +//! +//! **The two sides reach the hardware by different paths**, which is what makes this a test rather +//! than a tautology: IDF's side goes through `src/oracle/intr_ref.c` into IDF's own inlines and +//! macros, ours goes through `src/hal/intr.zig`, and the harness compares raw words rather than +//! either side's read-back. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +const intr = hal.intr; +const Reg = mmio.Reg; +const Field = mmio.Field; + +// ---------------------------------------------------------------- ESP-IDF's side + +extern fn oracle_intr_intthresh_standard() c_int; +extern fn oracle_intr_mintstatus_csr() c_int; +extern fn oracle_intr_mtvt_csr() c_int; +extern fn oracle_intr_nlbits() c_int; +extern fn oracle_intr_ext_offset() c_int; +extern fn oracle_intr_thresh_reg_addr() c_uint; +extern fn oracle_intr_ctrl_reg_addr(clic_id: c_uint) c_uint; + +extern fn oracle_intr_route(intr_src: c_uint, line: c_uint) void; +extern fn oracle_intr_unroute(intr_src: c_uint) void; +extern fn oracle_intr_set_vectored(line: c_uint, vectored: c_int) void; +extern fn oracle_intr_get_type(line: c_uint) c_int; +extern fn oracle_intr_get_priority(line: c_uint) c_int; +extern fn oracle_intr_enable(line: c_uint) void; +extern fn oracle_intr_disable(line: c_uint) void; +extern fn oracle_intr_set_type(line: c_uint, trig: c_uint) void; +extern fn oracle_intr_set_priority(line: c_uint, priority: c_uint) void; +extern fn oracle_intr_edge_ack(line: c_uint) void; +extern fn oracle_intr_enabled_mask() c_uint; +extern fn oracle_intr_set_threshold(level: c_uint) void; +extern fn oracle_intr_get_threshold() c_uint; +extern fn oracle_intr_set_mtvt(mtvt: c_uint) void; + +/// The constants ESP-IDF's half was compiled with. Every one of these is a way the experiment could +/// be quietly meaningless, so the harness should print them rather than assume them: +/// * `intthresh_standard` **must be 0**. A 1 means the C side switched to the `mintthresh` CSR, +/// which this die does not implement, and the threshold comparison would be against a register +/// the interrupt arbiter never reads. (`intr_ref.c` also makes this a compile error.) +/// * `mintstatus_csr` must be 0x346 - the non-standard number for pre-v3 silicon. 0xFB1 would mean +/// the rev-3 header path got selected. +/// * `ext_offset` must be 16 and `nlbits` 3: all the arithmetic on both sides rests on those two. +/// * `thresh_reg` must be 0x20800008, not a CSR number. +pub const RefConfig = struct { + intthresh_standard: u32, + mintstatus_csr: u32, + mtvt_csr: u32, + nlbits: u32, + ext_offset: u32, + thresh_reg: u32, +}; + +pub fn refConfig() RefConfig { + return .{ + .intthresh_standard = @intCast(oracle_intr_intthresh_standard()), + .mintstatus_csr = @intCast(oracle_intr_mintstatus_csr()), + .mtvt_csr = @intCast(oracle_intr_mtvt_csr()), + .nlbits = @intCast(oracle_intr_nlbits()), + .ext_offset = @intCast(oracle_intr_ext_offset()), + .thresh_reg = @intCast(oracle_intr_thresh_reg_addr()), + }; +} + +// ---------------------------------------------------------------- what is under test + +/// The source under test. A module-level variable because Zig has no closures and the harness holds +/// plain `fn` pointers - the same reason `gpio_cases.zig:39` has one. +pub var source: intr.Source = .tg1_t0; + +/// The external line under test, 0..31. +pub var line: u5 = 5; + +/// Sources worth routing, chosen to exercise the address arithmetic rather than to be interesting: +/// the first mapping register (`lp_rtc`, +0x000), the last one that exists on this die +/// (`assist_debug`, +0x1FC), and three in between. A wrong scale factor or a wrong base would be +/// invisible at `lp_rtc` and unmissable at `assist_debug`. +/// +/// The three rev-3-only sources (IDs 133-135) are deliberately absent: their mapping registers are +/// not implemented on pre-v3 silicon, so a comparison there would compare two reads of nothing. +pub const sources = [_]intr.Source{ .lp_rtc, .i2c1, .tg1_t0, .gpio_intr3, .assist_debug }; + +/// Lines worth testing: one in the low half, one in the high half. A line is offset by 16 before it +/// indexes the CLIC, so line 24 lands at CLIC ID 40 - past the point where a missing offset would +/// still have landed inside the 48-word table and gone unnoticed. +pub const lines = [_]u5{ 5, 24 }; + +// ---------------------------------------------------------------- addresses + +const matrix_base: u32 = @intCast(regs.DR_REG_INTERRUPT_CORE0_BASE); +const clic_ctrl_base: u32 = @intCast(regs.DR_REG_CLIC_CTRL_BASE); + +const int_map = Field.of(regs.INTERRUPT_CORE0_UART0_INT_MAP_S, regs.INTERRUPT_CORE0_UART0_INT_MAP_V); +const int_ctl = Field.of(regs.CLIC_INT_CTL_S, regs.CLIC_INT_CTL_V); +const int_attr_trig = Field.of(regs.CLIC_INT_ATTR_TRIG_S, regs.CLIC_INT_ATTR_TRIG_V); +const int_attr_shv = Field.of(regs.CLIC_INT_ATTR_SHV_S, regs.CLIC_INT_ATTR_SHV_V); +const int_ie = Field.of(regs.CLIC_INT_IE_S, regs.CLIC_INT_IE_V); + +inline fn mapReg(source_id: u8) Reg { + return Reg.atAddress(matrix_base + 4 * @as(u32, source_id)); +} +inline fn ctrlReg(clic_id: u32) Reg { + return Reg.atAddress(clic_ctrl_base + 4 * clic_id); +} + +// ---------------------------------------------------------------- restore + +/// Reset value of the CLIC_INT_CTL priority field: 0x1f (`soc/clic_reg.h:70`). Restoring to 0 would +/// be restoring to a state the hardware never boots in, and the two implementations would then be +/// compared from a starting point neither of them produces. +const ctl_reset: u32 = 0x1f; + +/// Bring all three blocks back to a known state between the two implementations. +/// +/// Configuring rather than resetting, and here that is not a preference: neither the matrix nor the +/// CLIC has a reset bit in HP_SYS_CLKRST, and there is no sound way for code to reset the interrupt +/// controller of the core it is running on. Configuring is legitimate because every field this +/// touches is plain read/write. +/// +/// Three things this does that a narrower restore would not, each for a reason: +/// +/// 1. **All five sources, not just the current one.** A case that wrote the wrong mapping register +/// would otherwise leave that write behind; the next case's two snapshots would both inherit it +/// and compare equal. The bug would be visible exactly once and then absorbed. +/// +/// 2. **A sweep of all 128 mapping registers for anything pointing at a line under test.** This is +/// what makes "nothing can assert into these lines during the run" true rather than hoped. The +/// ROM bootloader is under no obligation to leave the matrix clear, and a live peripheral routed +/// to CLIC ID 21 or 40 would set that line's pending bit between the two snapshots and read as a +/// false difference. 128 reads is nothing; guessing is not free. +/// +/// 3. **Both lines, not just the current one**, for the same reason as (1). +fn restore() void { + for (sources) |s| intr.unroute(s); + + // (2): detach anything at all that aims at a line this suite uses. + for (lines) |l| { + const id = @as(u32, l) + intr.ext_offset; + var src: u32 = 0; + while (src <= intr.max_source_id) : (src += 1) { + const r = mapReg(@intCast(src)); + if (r.get(int_map) == id) r.modify(.{int_map.is(0)}); + } + } + + for (lines) |l| { + ctrlReg(@as(u32, l) + intr.ext_offset).modify(.{ + int_ctl.is(ctl_reset), + int_attr_trig.is(0), + int_attr_shv.is(0), + int_ie.is(0), + }); + } + + intr.setThreshold(0); +} + +/// Run once, before anything. Makes the "no interrupt can be taken during this suite" claim true +/// rather than assumed: the cases enable CLIC lines, and an enabled line with MIE set would vector +/// through whatever mtvec the bootloader happened to leave behind. +fn setup() void { + intr.globalDisable(); + restore(); +} + +// ---------------------------------------------------------------- the matrix suite + +/// The interrupt matrix: 128 mapping registers, source 0 at +0x000 through `assist_debug` at +/// +0x1FC. The whole block is in the window deliberately - the cases touch five of the 128, and the +/// other 123 are the point: a routing write that landed on the wrong register shows up as a +/// difference in a word no case names. +/// +/// No `volatile_words`. Each word is a 6-bit read/write field plus reserved bits; nothing here is +/// read-to-clear, nothing self-clears, and no hardware writes these - they are pure configuration. +/// +/// No `clock` either, and that is structural rather than lucky: the matrix and the CLIC are in the +/// CPU's own clock domain and have no gate in HP_SYS_CLKRST, because a core cannot be allowed to +/// gate off the block that delivers its own interrupts. +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "intr_matrix", + .base = matrix_base, + .words = 128, + .restore = .{ .configure = restore }, + }, + .setup = setup, + .cases = &.{ + .{ .name = "route", .idf = idfRoute, .ours = ourRoute }, + .{ .name = "unroute", .idf = idfUnroute, .ours = ourUnroute }, + .{ .name = "route_rewrite", .idf = idfRouteTwice, .ours = ourRouteTwice }, + .{ .name = "route_preserves_reserved", .idf = idfRouteOverJunk, .ours = ourRouteOverJunk }, + }, +}; + +// ---------------------------------------------------------------- the CLIC control suite + +/// The CLIC's per-interrupt control file: 48 words, one per CLIC ID, the 16 internal IDs included. +/// They are in the window on purpose - every accessor in `hal/intr.zig` adds 16 to the caller's line +/// number, and an implementation that forgot to would write into IDs 5 and 24 instead of 21 and 40. +/// Both of those are inside this window and neither is excluded below, so a missing offset is a +/// visible difference rather than silence. +/// +/// **The pending bits, and why only three words are excluded.** CLIC_INT_IP is bit 0 of every one of +/// these words and the hardware sets it on its own when a source asserts. For the 32 external IDs +/// that cannot happen during this run: `restore` sweeps all 128 mapping registers and detaches +/// anything aimed at a line under test, and the other 30 external lines have nothing routed to them +/// that these cases did not route. The three excluded words are the standard RISC-V internal +/// interrupts, whose pending bits are driven by the core's own timer and software-interrupt +/// hardware rather than by the matrix, and which this file therefore cannot promise are quiet. +/// Excluding them costs nothing: no operation here can reach an internal ID except by the off-by-16 +/// bug, and that bug lands on IDs 5 and 24, which are still compared. +pub const clic_suite: types.Suite = .{ + .descriptor = .{ + .name = "intr_clic", + .base = clic_ctrl_base, + .words = 48, + .volatile_words = &.{ + 3, // machine software interrupt - IP driven by the msip mechanism + 7, // machine timer interrupt - IP driven by the core timer, which is running + 11, // machine external interrupt - IP driven from outside the matrix + }, + .restore = .{ .configure = restore }, + }, + .setup = setup, + .cases = &.{ + .{ .name = "enable", .arg = 1, .idf = idfEnable, .ours = ourEnable }, + .{ .name = "disable", .arg = 0, .idf = idfDisable, .ours = ourDisable }, + .{ .name = "enable_other_line", .idf = idfEnableOther, .ours = ourEnableOther }, + .{ .name = "trigger_level", .arg = 0, .idf = idfTrigLevel, .ours = ourTrigLevel }, + .{ .name = "trigger_rising", .arg = 1, .idf = idfTrigRising, .ours = ourTrigRising }, + .{ .name = "trigger_falling", .arg = 3, .idf = idfTrigFalling, .ours = ourTrigFalling }, + .{ .name = "priority", .arg = 7, .idf = idfPrio7, .ours = ourPrio7 }, + .{ .name = "priority", .arg = 1, .idf = idfPrio1, .ours = ourPrio1 }, + .{ .name = "priority", .arg = 0, .idf = idfPrio0, .ours = ourPrio0 }, + .{ .name = "vectored_on", .arg = 1, .idf = idfVectoredOn, .ours = ourVectoredOn }, + .{ .name = "vectored_off", .arg = 0, .idf = idfVectoredOff, .ours = ourVectoredOff }, + .{ .name = "edge_ack", .idf = idfEdgeAck, .ours = ourEdgeAck }, + .{ .name = "configure_line", .arg = 3, .idf = idfConfigure, .ours = ourConfigure }, + }, +}; + +// ---------------------------------------------------------------- the threshold suite + +/// The CLIC's three global registers, and the reason this file exists at all. +/// +/// 0x2080_0000 CLIC_INT_CONFIG (R/W in MNLBITS, untouched here), 0x2080_0004 CLIC_INT_INFO (RO, +/// reads 48 interrupts / 4 CTL bits), 0x2080_0008 CLIC_INT_THRESH. Three words, contiguous, all +/// mapped - a window of exactly the registers involved, rather than a widened one that reaches the +/// threshold by crossing 4 KiB of nothing. +/// +/// **This is the die where the threshold is a register and not a CSR.** `soc/interrupt_reg.h:28-40` +/// sets INTTHRESH_STANDARD 0 under CONFIG_ESP32P4_SELECTS_REV_LESS_V3, and `riscv/csr_clic.h:37-47` +/// then never defines MINTTHRESH_CSR. A HAL that wrote CSR 0x347 instead would pass every other +/// case in this file and fail every one of these three, which is exactly the discrimination the +/// suite is for: the mistake is silent everywhere else. +pub const thresh_suite: types.Suite = .{ + .descriptor = .{ + .name = "intr_thresh", + .base = @intCast(regs.DR_REG_CLIC_BASE), + .words = 3, + .restore = .{ .configure = restoreThreshold }, + }, + .setup = setup, + .cases = &.{ + .{ .name = "threshold", .arg = 0, .idf = idfThresh0, .ours = ourThresh0 }, + .{ .name = "threshold", .arg = 3, .idf = idfThresh3, .ours = ourThresh3 }, + .{ .name = "threshold", .arg = 7, .idf = idfThresh7, .ours = ourThresh7 }, + }, +}; + + +/// Restored through ESP-IDF's side, never through the code under test. `differ.zig` runs restore, +/// idf, snapshot, restore, ours, snapshot: with the HAL on both the restore and the "ours" side, a +/// HAL function that does nothing leaves run B's snapshot equal to run A's and the case passes. That +/// makes a suite blind to precisely the failure it was written to catch. +fn restoreThreshold() void { + oracle_intr_set_threshold(0); +} + +// ---------------------------------------------------------------- matrix cases + +fn idfRoute() void { + oracle_intr_route(@intFromEnum(source), line); +} +fn ourRoute() void { + intr.route(source, line); +} + +fn idfUnroute() void { + oracle_intr_route(@intFromEnum(source), line); + oracle_intr_unroute(@intFromEnum(source)); +} +fn ourUnroute() void { + intr.route(source, line); + intr.unroute(source); +} + +/// Re-routing a source that is already routed. The mapping register is a read-modify-write of the +/// low 6 bits (`interrupt_clic_ll.h:46`, RV_INT_MASK 63 at line 25), so the second write must +/// *replace* the first rather than OR into it. An `|=` implementation passes the single-write case +/// and fails this one: 31+16 = 47 or'd with 5+16 = 21 is 63, not 21. +fn idfRouteTwice() void { + oracle_intr_route(@intFromEnum(source), 31); + oracle_intr_route(@intFromEnum(source), line); +} +fn ourRouteTwice() void { + intr.route(source, 31); + intr.route(source, line); +} + +/// Route over a word whose reserved bits [31:6] are all set. Both sides must preserve them - IDF's +/// REG_SET_BITS masks with 63, ours is a `Field` of the same width - and this is the case that says +/// so rather than assuming it. A `write` where a `modify` belonged clears them. +fn dirtyMapReg() void { + mapReg(@intFromEnum(source)).writeRaw(0xffff_ffc0); +} +fn idfRouteOverJunk() void { + dirtyMapReg(); + oracle_intr_route(@intFromEnum(source), line); +} +fn ourRouteOverJunk() void { + dirtyMapReg(); + intr.route(source, line); +} + +// ---------------------------------------------------------------- CLIC control cases + +fn idfEnable() void { + oracle_intr_enable(line); +} +fn ourEnable() void { + intr.setEnabled(line, true); +} + +fn idfDisable() void { + oracle_intr_enable(line); + oracle_intr_disable(line); +} +fn ourDisable() void { + intr.setEnabled(line, true); + intr.setEnabled(line, false); +} + +fn otherLine() u5 { + return if (line == lines[0]) lines[1] else lines[0]; +} + +/// Enable the line under test and then enable and disable the *other* one. Catches an index bug +/// that a single-line case cannot: both lines are inside the window, so touching the wrong one is a +/// visible difference rather than an invisible no-op, and the enable/disable pair means the correct +/// answer is "only the first line ends up enabled". +fn idfEnableOther() void { + oracle_intr_enable(line); + oracle_intr_enable(otherLine()); + oracle_intr_disable(otherLine()); +} +fn ourEnableOther() void { + intr.setEnabled(line, true); + intr.setEnabled(otherLine(), true); + intr.setEnabled(otherLine(), false); +} + +fn idfTrigLevel() void { + oracle_intr_set_type(line, 0); +} +fn ourTrigLevel() void { + intr.setTrigger(line, .level); +} + +fn idfTrigRising() void { + oracle_intr_set_type(line, 1); +} +fn ourTrigRising() void { + intr.setTrigger(line, .rising_edge); +} + +/// 0b11, falling edge. IDF's own non-ROM helper only ever writes rising - `esp_tee_rv_utils.h:97-98` +/// is a TODO saying as much - so this is an encoding IDF documents (`clic_reg.h:84-88`) but does not +/// exercise, and its reference is the transcribed REG_SET_FIELD from `esp_rom_clic.c:21` rather than +/// a call into IDF. That makes it the case here most likely to disagree, which is why it is here. +fn idfTrigFalling() void { + oracle_intr_set_type(line, 3); +} +fn ourTrigFalling() void { + intr.setTrigger(line, .falling_edge); +} + +fn idfPrio7() void { + oracle_intr_set_priority(line, 7); +} +fn ourPrio7() void { + intr.setPriority(line, 7); +} + +fn idfPrio1() void { + oracle_intr_set_priority(line, 1); +} +fn ourPrio1() void { + intr.setPriority(line, 1); +} + +/// Priority 0 writes 0x00 over the reset value 0x1f, so this is the case that proves the low +/// `8 - NLBITS` bits are being *cleared*. If either side padded them with ones - which is what the +/// *threshold* encoding does, `csr_clic.h:59` - the two words would differ by 0x1F000000 and by +/// nothing else. Priorities 1 and 7 both leave those bits zero either way and cannot see it. +fn idfPrio0() void { + oracle_intr_set_priority(line, 0); +} +fn ourPrio0() void { + intr.setPriority(line, 0); +} + +fn idfVectoredOn() void { + oracle_intr_set_vectored(line, 1); +} +fn ourVectoredOn() void { + intr.setVectored(line, true); +} + +fn idfVectoredOff() void { + oracle_intr_set_vectored(line, 1); + oracle_intr_set_vectored(line, 0); +} +fn ourVectoredOff() void { + intr.setVectored(line, true); + intr.setVectored(line, false); +} + +/// Writing 1 to IP. With nothing routed to this line there is nothing pending to clear, so what is +/// compared is the *store*: which word, which bit, and whether the surrounding fields survive. That +/// the hardware then reads that write as an acknowledgement is behavioural and out of reach here. +fn idfEdgeAck() void { + oracle_intr_set_type(line, 1); + oracle_intr_edge_ack(line); +} +fn ourEdgeAck() void { + intr.setTrigger(line, .rising_edge); + intr.edgeAck(line); +} + +/// The whole per-line configuration in one go. Our side goes through `configureLine`, not through +/// four separate calls, because that function - not its pieces - is what a driver will use, and +/// because four fields in one word is where an ordering bug or a `write` that should have been a +/// `modify` shows up while each field alone still passes. +fn idfConfigure() void { + oracle_intr_set_type(line, 1); + oracle_intr_set_priority(line, 3); + oracle_intr_set_vectored(line, 0); + oracle_intr_enable(line); +} +fn ourConfigure() void { + intr.configureLine(line, .{ + .handler = noopHandler, + .trigger = .rising_edge, + .priority = 3, + .vectored = false, + }); +} + +/// Installed and never called: `configureLine` requires a handler and the differ never sets MIE. +/// Its address goes into a RAM array no descriptor's window covers, so it cannot perturb a +/// comparison. +fn noopHandler(_: u5) void {} + +// ---------------------------------------------------------------- threshold cases + +fn idfThresh0() void { + oracle_intr_set_threshold(0); +} +fn ourThresh0() void { + intr.setThreshold(0); +} + +fn idfThresh3() void { + oracle_intr_set_threshold(3); +} +fn ourThresh3() void { + intr.setThreshold(3); +} + +fn idfThresh7() void { + oracle_intr_set_threshold(7); +} +fn ourThresh7() void { + intr.setThreshold(7); +} + +// --------------------------------------------------------------------------------------------- +// THE BEHAVIOURAL TEST - which the parent must run, because this file cannot. +// --------------------------------------------------------------------------------------------- +// +// Everything above compares register *state*. None of it touches the parts of this peripheral that +// exist only while an interrupt is in flight: mtvec's mode bits, the MTVT fetch, the CLIC's +// arbitration against the threshold, mcause's EXCCODE, the trap entry's register save, and `mret`. +// A HAL that passes every case above and still never delivers an interrupt is entirely possible - +// it is in fact the expected failure mode, because the single most likely mistake here (writing the +// `mintthresh` CSR instead of CLIC_INT_THRESH_REG) leaves no trace in any register. +// +// The cheap test, with the numbers it needs: +// +// 1. `hal.clkrst.init(.timg1)` - clock on, reset pulsed, flash-boot protection cleared. TIMG's +// reset re-arms the flash-boot watchdog; skipping the clear reboots the board a second later +// with nothing on the console to explain it. +// 2. Arm TIMG1 timer 0 for a one-shot alarm a few milliseconds out, alarm enabled, and the +// timer's own interrupt enable set (TIMG_T0_INT_ENA). +// 3. `hal.intr.init()` - fills the 48-entry vector table with the trap entry, writes MTVT +// (CSR 0x307), writes mtvec = trap_entry | 3, and opens the threshold to 0. +// 4. `hal.intr.attach(.tg1_t0, 5, .{ .handler = h, .trigger = .level, .priority = 1 })`. +// `.tg1_t0` is source ID 49, so the mapping register is DR_REG_INTERRUPT_CORE0_BASE + 0xC4 and +// the value written is 5 + 16 = 21. Priority 1 against threshold 0 is the minimum that is not +// masked: the threshold comparison is inclusive, so priority 0 would never fire. +// 5. The handler increments a counter and **clears TIMG1's interrupt status**. That is mandatory +// for a level source: the CLIC has no acknowledge for level, so a handler that returns without +// clearing the peripheral re-enters immediately and the board sits inside the trap entry with +// the console silent. That failure looks exactly like a crash and is not one. +// 6. `hal.intr.globalEnable()`, spin ~50 ms, `hal.intr.globalDisable()`. +// +// Pass is `counter == 1` **and** `hal.intr.spurious == 0`. Both halves matter: a counter of 1 with a +// non-zero spurious count means an interrupt also arrived on a line nobody claimed, i.e. a matrix +// write went somewhere unintended. +// +// Diagnostics worth printing on failure, because they separate the ways this can go wrong: +// * `hal.intr.getThreshold()` beside `oracle_intr_get_threshold()` - a disagreement means the +// threshold mechanism is the fault, which is what this die's non-standard CLIC invites. +// * `hal.intr.routedLine(.tg1_t0)` - null means the matrix write missed. +// * `hal.intr.isPending(5)` with the counter at 0 - the CLIC latched it and the core never took +// it, so the fault is mtvec, MTVT or MIE, and is neither the matrix nor the threshold. +// * TIMG1's raw interrupt status - if that is 0 the timer never fired and the test is measuring +// something else entirely. +// +// A second, sharper test once the first passes: set the threshold to 7 *before* enabling, confirm +// the counter stays 0 while `isPending(5)` becomes 1, then drop the threshold to 0 and confirm the +// pending interrupt is delivered. That is the only way to show the memory-mapped threshold register +// is the one the arbiter actually reads, and it is the claim this whole file is least able to +// support on its own. + +// --------------------------------------------------------------------------------------------- + +test "the line-to-CLIC-ID offset the cases assume is the one the header defines" { + try std.testing.expectEqual(@as(u32, 16), intr.ext_offset); + try std.testing.expectEqual(@as(u32, 48), intr.total_ids); + // Both test lines land inside the 48-word CLIC window, which is what makes an off-by-16 in + // hal/intr.zig visible to the harness rather than silent... + for (lines) |l| try std.testing.expect(@as(u32, l) + intr.ext_offset < intr.total_ids); + // ...and neither un-offset line is one of the three words excluded as volatile, or the bug + // would land in a word the harness ignores. + for (lines) |l| for (clic_suite.descriptor.volatile_words) |w| try std.testing.expect(w != l); +} + +test "the matrix window covers every source the cases route" { + for (sources) |s| { + try std.testing.expect(!s.isRev3Only()); + try std.testing.expect(@intFromEnum(s) < suite.descriptor.words); + } + // The extremes really are in the set - that is the point of the choice. + try std.testing.expectEqual(@as(u8, 0), @intFromEnum(sources[0])); + try std.testing.expectEqual(@as(u8, 127), @intFromEnum(sources[sources.len - 1])); +} + +test "the threshold window is the register block, not a CSR, and holds all three words" { + try std.testing.expectEqual(@as(u32, 0x2080_0000), thresh_suite.descriptor.base); + // CLIC_INT_THRESH_REG is the third word. If this ever stops being true the window is wrong. + try std.testing.expectEqual( + @as(u32, 0x2080_0008), + thresh_suite.descriptor.base + 4 * (thresh_suite.descriptor.words - 1), + ); +} diff --git a/src/oracle/intr_ref.c b/src/oracle/intr_ref.c new file mode 100644 index 0000000..ceeaa0d --- /dev/null +++ b/src/oracle/intr_ref.c @@ -0,0 +1,211 @@ +/* ESP-IDF's own CLIC code, given external linkage so the differential harness can call it. + * + * The interrupt controller is the one peripheral where "wrap IDF's LL header" is not the whole + * story, and the reason is worth recording rather than papering over. + * + * ESP-IDF splits CLIC access across four places: + * 1. components/hal/include/hal/interrupt_clic_ll.h - the matrix route, SHV, and the two + * getters. Included below and wrapped directly; this is the LL proper. + * 2. components/riscv/include/esp_private/interrupt_clic.h - MTVT, the threshold, edge-ack and + * the enabled-mask scan, all `FORCE_INLINE_ATTR`. Also included below and wrapped directly. + * 3. the **mask ROM** - esprv_intc_int_enable / _set_priority / _set_type, aliased into + * esprv_int_* by components/riscv/ld/rom.api.ld. There is no C source for these, so they + * cannot be compiled into this image as a reference. Worse, one of them is *wrong* on this + * die: components/esp_rom/patches/esp_rom_clic.c:12-22 exists because the ROM's + * esprv_intc_int_set_type silently configures LEVEL when asked for EDGE, on exactly the + * CONFIG_ESP32P4_SELECTS_REV_LESS_V3 silicon this board is. + * 4. components/esp_tee/.../clic/esp_tee_rv_utils.h - a non-ROM implementation of enable, + * disable, set_type and set_priority, written as byte stores. + * + * For (3) the reference below is the register expression from IDF's own replacement code, copied + * statement for statement with the file and line it came from, and using IDF's macros so the + * numbers are still IDF's. That is a transcription, and it is the weakest link in this file; it is + * marked as such at each site. Everything else calls IDF's code directly. + * + * Note also what the ROM situation means for the differential's *value* here: for enable, priority + * and trigger the comparison is against IDF's non-ROM path, which is the path IDF itself uses on + * TEE builds and the path its ROM patch restores. It is not against the ROM function a stock + * app_main would reach. + */ + +/* IDF's clock and reset LL functions are shadowed by a wrapper macro referencing + * `__DECLARE_RCC_ATOMIC_ENV`, an identifier IDF never defines anywhere, so that an unguarded call + * fails to compile. Nothing in this translation unit gates a clock, but the header chain reaches + * those declarations, so the name has to exist. Same reasoning as src/oracle/gpio_ref.c:18. */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); + +/* **First, and load-bearing.** soc/interrupt_reg.h tests CONFIG_ESP32P4_SELECTS_REV_LESS_V3 but + * does not include sdkconfig.h itself - it relies on the caller having done so, which in IDF's own + * build happens because CMake force-includes it. Include it *after* any header below and + * INTTHRESH_STANDARD comes out 1, the rev-3 answer, and this reference would be built against the + * mintthresh CSR that this silicon does not implement. That is not hypothetical: this file's first + * version had the includes in the obvious order and the #error below fired. + * + * build.zig now also passes `-include oracle_sdkconfig.h` to every reference translation unit, so + * this line is belt as well as braces. It stays because the ordering constraint is a property of + * IDF's headers, not of our build flags, and the next person to reorder these should see why. */ +#include "sdkconfig.h" +#include "soc/soc.h" +#include "soc/clic_reg.h" +#include "soc/interrupt_reg.h" +#include "hal/interrupt_clic_ll.h" +#include "esp_private/interrupt_clic.h" + +/* Guard the whole point of this file: if the build ever stopped defining + * CONFIG_ESP32P4_SELECTS_REV_LESS_V3 (src/oracle/oracle_sdkconfig.h:25), interrupt_reg.h:28-40 would + * flip INTTHRESH_STANDARD to 1 and every threshold function below would silently switch from the + * memory-mapped register to the mintthresh CSR - which this die does not implement. The reference + * would then be comparing against a threshold mechanism that does not exist, and would agree with + * nothing. Fail the compile instead. */ +#if INTTHRESH_STANDARD +#error "this die uses the memory-mapped CLIC threshold; INTTHRESH_STANDARD must be 0 here" +#endif + +/* Report the numbers this reference was compiled with, so a run can never silently be against the + * wrong variant of the controller. */ +int oracle_intr_intthresh_standard(void) +{ + return INTTHRESH_STANDARD; +} + +int oracle_intr_mintstatus_csr(void) +{ + return MINTSTATUS_CSR; +} + +int oracle_intr_mtvt_csr(void) +{ + return MTVT_CSR; +} + +int oracle_intr_nlbits(void) +{ + return NLBITS; +} + +int oracle_intr_ext_offset(void) +{ + return CLIC_EXT_INTR_NUM_OFFSET; +} + +unsigned oracle_intr_thresh_reg_addr(void) +{ + return (unsigned)CLIC_INT_THRESH_REG; +} + +unsigned oracle_intr_ctrl_reg_addr(unsigned clic_id) +{ + return (unsigned)CLIC_INT_CTRL_REG(clic_id); +} + +/* ---------------------------------------------------------------- the interrupt matrix */ + +/* interrupt_clic_ll.h:35-48, with the `+ RV_EXTERNAL_INT_OFFSET` that riscv/interrupt_clic.c:26 + * applies before calling it. Core 0 only: this image never releases core 1. */ +void oracle_intr_route(unsigned intr_src, unsigned line) +{ + interrupt_clic_ll_route(0, (int)intr_src, (int)line + RV_EXTERNAL_INT_OFFSET); +} + +/* esp_system/port/cpu_start.c:185 - IDF's own way to detach a source, writing ETS_INVALID_INUM + * (0 on this chip, soc/esp32p4/include/soc/soc.h:251) with no offset added. */ +void oracle_intr_unroute(unsigned intr_src) +{ + interrupt_clic_ll_route(0, (int)intr_src, ETS_INVALID_INUM); +} + +/* ---------------------------------------------------------------- per-line control */ + +/* interrupt_clic_ll.h:99-102 via riscv/interrupt_clic.c:48-51. */ +void oracle_intr_set_vectored(unsigned line, int vectored) +{ + interrupt_clic_ll_set_vectored((int)line + RV_EXTERNAL_INT_OFFSET, vectored != 0); +} + +/* interrupt_clic_ll.h:58-61 via riscv/interrupt_clic.c:30-33: 1 for edge, 0 for level. */ +int oracle_intr_get_type(unsigned line) +{ + return interrupt_clic_ll_get_type((int)line + RV_EXTERNAL_INT_OFFSET); +} + +/* interrupt_clic_ll.h:71-75 via riscv/interrupt_clic.c:36-39. */ +int oracle_intr_get_priority(unsigned line) +{ + return interrupt_clic_ll_get_priority((int)line + RV_EXTERNAL_INT_OFFSET); +} + +/* TRANSCRIBED, not called: the ROM owns esprv_intc_int_enable and there is no source for it. + * The store is esp_tee/subproject/main/include/clic/esp_tee_rv_utils.h:74, verbatim - a byte write + * of BYTE_CLIC_INT_IE to BYTE_CLIC_INT_IE_REG. Byte 1 of the control word holds nothing but IE, so + * this and a 32-bit read-modify-write of CLIC_INT_IE leave the same word behind; that equivalence + * is precisely what the differential is there to check rather than assert. */ +void oracle_intr_enable(unsigned line) +{ + const unsigned id = line + CLIC_EXT_INTR_NUM_OFFSET; + *(uint8_t volatile *)(BYTE_CLIC_INT_IE_REG(id)) = BYTE_CLIC_INT_IE; +} + +/* TRANSCRIBED: esp_tee_rv_utils.h:88. */ +void oracle_intr_disable(unsigned line) +{ + const unsigned id = line + CLIC_EXT_INTR_NUM_OFFSET; + *(uint8_t volatile *)(BYTE_CLIC_INT_IE_REG(id)) = 0; +} + +/* TRANSCRIBED: esp_rom/patches/esp_rom_clic.c:21, which is IDF's *replacement* for the ROM's + * broken esprv_intc_int_set_type on pre-v3 P4 silicon. A 32-bit REG_SET_FIELD on CLIC_INT_ATTR_TRIG, + * so unlike the TEE build's byte store it preserves SHV by read-modify-write rather than by the + * byte's other bits happening to be reloaded - same result, different mechanism. `type` is the raw + * two-bit encoding (0 level, 1 rising, 3 falling; clic_reg.h:84-88). */ +void oracle_intr_set_type(unsigned line, unsigned type) +{ + const unsigned id = line + CLIC_EXT_INTR_NUM_OFFSET; + REG_SET_FIELD(CLIC_INT_CTRL_REG(id), CLIC_INT_ATTR_TRIG, type); +} + +/* TRANSCRIBED: esp_tee_rv_utils.h:112. Note the encoding - priority left-aligned into the top + * NLBITS of the byte with the low bits **zero**, which differs from the threshold's encoding + * below. */ +void oracle_intr_set_priority(unsigned line, unsigned priority) +{ + const unsigned id = line + CLIC_EXT_INTR_NUM_OFFSET; + *(uint8_t volatile *)(BYTE_CLIC_INT_CTL_REG(id)) = (uint8_t)(priority << BYTE_CLIC_INT_CTL_S); +} + +/* esp_private/interrupt_clic.h, rv_utils_intr_edge_ack: writing 1 to IP is what *clears* an + * edge-triggered pending. Called directly - this one is a real IDF inline. */ +void oracle_intr_edge_ack(unsigned line) +{ + rv_utils_intr_edge_ack(line); +} + +/* esp_private/interrupt_clic.h, rv_utils_intr_get_enabled_mask. */ +unsigned oracle_intr_enabled_mask(void) +{ + return rv_utils_intr_get_enabled_mask(); +} + +/* ---------------------------------------------------------------- the threshold */ + +/* esp_private/interrupt_clic.h:153-156 -> :129-146. Called directly, so the reference includes + * IDF's own read-back-to-force-the-store and IDF's own NLBITS_TO_BYTE padding, and the harness + * compares against those rather than against a re-derivation of them. */ +void oracle_intr_set_threshold(unsigned level) +{ + rv_utils_restore_intlevel(level); +} + +/* esp_private/interrupt_clic.h:44-57. Returns an absolute level 0..7. */ +unsigned oracle_intr_get_threshold(void) +{ + return rv_utils_get_interrupt_threshold(); +} + +/* ---------------------------------------------------------------- vector table */ + +/* esp_private/interrupt_clic.h:63-66. MTVT is CSR 0x307. Writing it has no effect on any register + * the harness photographs, so this exists for the behavioural test rather than for the diff. */ +void oracle_intr_set_mtvt(unsigned mtvt) +{ + rv_utils_set_mtvt(mtvt); +} diff --git a/src/oracle/ledc_cases.zig b/src/oracle/ledc_cases.zig new file mode 100644 index 0000000..5a39f83 --- /dev/null +++ b/src/oracle/ledc_cases.zig @@ -0,0 +1,513 @@ +//! LEDC's side of the differential test: every operation expressed as ESP-IDF's LL calls and as this +//! project's HAL calls. +//! +//! **A register diff cannot prove that a commit happened.** `LEDC_PARA_UP_CHn` and +//! `LEDC_TIMERn_PARA_UP` are write-to-trigger bits that the hardware clears again by itself, so the +//! word that carried the commit reads back exactly as it did before, and the shadow registers the +//! commit copies into are not addressable. Two snapshots therefore agree whether or not either +//! implementation committed anything at all. Nothing in this file claims otherwise. +//! +//! What the diff *can* prove, and what these cases are shaped to prove: +//! +//! * The **staged values** match. Every case stages through the same fields IDF's LL stages, so a +//! wrong shift, a wrong instance stride or a `write` where a `modify` was needed shows up in the +//! staged word - which is the register the commit will read. +//! * The commit **did not destroy the staging**. This is the real hazard of a commit bit that lives +//! inside the word it commits: `LEDC_PARA_UP_CH0` is bit 4 of `LEDC_CH0_CONF0_REG`, so a commit +//! implemented as `writeRaw(1 << 4)` would zero `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV` and +//! `OVF_NUM` on its way past. That failure is loud here: the staged word would differ. +//! * `stage_without_commit` pins the distinction down. It stages a duty and stops, on both sides. +//! It must pass, and it must pass for the same reason a committed case passes - which is the +//! evidence that "passes" says nothing about the commit. +//! +//! `LEDC_CHn_DUTY_R_REG` is the one register that reflects the committed shadow rather than the +//! staged value, and it is listed as volatile below rather than used as proof: it updates when the +//! timer next overflows, so what it holds at snapshot time depends on where the counter happened to +//! be. Proving the commit needs an oscilloscope, or the ovf-count interrupt, not a register read. +//! +//! Four windows, because LEDC's state is not in one place: the peripheral block, its gamma RAM +//! aperture, the GPIO matrix (pin routing touches no LEDC register at all) and HP_SYS_CLKRST (where +//! the P4 moved LEDC's clock mux). One suite each, since a `Peripheral` descriptor is one contiguous +//! range of words. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +const ledc = hal.ledc; + +extern fn oracle_ledc_enable_function_clock(enable: c_int) void; +extern fn oracle_ledc_set_clock_source(sel: c_uint) void; +extern fn oracle_ledc_divisor(src_clk_freq: c_uint, freq_hz: c_int, precision: c_uint) c_uint; +extern fn oracle_ledc_set_clock_divider(timer: c_uint, div: c_uint) void; +extern fn oracle_ledc_set_duty_resolution(timer: c_uint, bits: c_uint) void; +extern fn oracle_ledc_commit_timer(timer: c_uint) void; +extern fn oracle_ledc_reset_timer(timer: c_uint) void; +extern fn oracle_ledc_pause_timer(timer: c_uint) void; +extern fn oracle_ledc_resume_timer(timer: c_uint) void; +extern fn oracle_ledc_configure_timer(timer: c_uint, src_hz: c_uint, freq_hz: c_int, resolution: c_uint) void; +extern fn oracle_ledc_bind_timer(channel: c_uint, timer: c_uint) void; +extern fn oracle_ledc_set_hpoint(channel: c_uint, hpoint: c_uint) void; +extern fn oracle_ledc_set_duty(channel: c_uint, duty: c_uint) void; +extern fn oracle_ledc_set_output_enabled(channel: c_uint, enable: c_int) void; +extern fn oracle_ledc_set_idle_level(channel: c_uint, level: c_uint) void; +extern fn oracle_ledc_commit_channel(channel: c_uint) void; +extern fn oracle_ledc_start(channel: c_uint) void; +extern fn oracle_ledc_stop(channel: c_uint, idle_level: c_uint) void; +extern fn oracle_ledc_configure_channel( + channel: c_uint, + timer: c_uint, + duty: c_uint, + hpoint: c_uint, + idle_level: c_uint, + output_enabled: c_int, +) void; +extern fn oracle_ledc_set_pin(pin: c_uint, channel: c_uint) void; + +/// The channel, timer and pad under test. Module-level variables because Zig has no closures and the +/// harness stores plain `fn` pointers; the alternative, a comptime-specialised pair per channel, +/// would compare code this project does not ship. +/// +/// The suite is safe to run once per pair, the way GPIO's is run once per pin - `channels` and +/// `timers` name the pairs worth using: instance 0, and the far end of each range, where a wrong +/// `RegArray` stride would land outside the block. +pub var channel: u32 = 0; +pub var timer: u32 = 0; +/// GPIO33 is a free pin on this board's JP1 header. GPIO20 is the LED, which the harness itself +/// leaves blinking, and GPIO54 is the ESP32-C6's reset line and must never be driven. +pub var pin: u8 = 33; + +pub const channels = [_]u32{ 0, 7 }; +pub const timers = [_]u32{ 0, 3 }; + +/// 40 MHz XTAL: `ClockSource.xtal.hz()`, and what `setup` selects. Passed explicitly to both sides +/// so the two arithmetics are compared on the same input rather than on each side's idea of the +/// clock tree. +const src_hz: u32 = ledc.xtal_hz; + +/// Bring LEDC up before the first case: its APB gate is off at power-on, so without this every +/// snapshot would be the last value the bus latched and the harness would (correctly) skip the whole +/// suite on the `clock` check. +fn setup() void { + ledc.init(.xtal); +} + +// ------------------------------------------------------------------- the peripheral block itself + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "ledc", + .base = @intCast(regs.LEDC_CH0_CONF0_REG), + // 96 words, 0x000-0x17f: eight channels (0x000-0x09f), four timers (0x0a0-0x0bf), the + // interrupt registers, the per-channel gamma *configuration* at 0x100-0x11f (the range + // count lives there, and `setDuty` writes it), the ETM enables, the timer compare and + // capture registers, and LEDC_CONF/LEDC_DATE at 0x170/0x174. Wide enough that every + // register any operation in this file touches is inside it except the gamma RAM aperture at + // 0x400, which has its own suite below. + // + // The reserved gaps (0x0d0-0x0ff, 0x130-0x13f, 0x160-0x16f) are read as well, deliberately: + // if a reserved word does not read back stably the diff will name the offset instead of + // hiding it. + .words = 96, + .volatile_words = &.{ + // LEDC_CHn_DUTY_R: the committed duty shadow, reloaded on timer overflow. + (0x010 - 0x000) / 4, (0x024 - 0x000) / 4, (0x038 - 0x000) / 4, (0x04c - 0x000) / 4, + (0x060 - 0x000) / 4, (0x074 - 0x000) / 4, (0x088 - 0x000) / 4, (0x09c - 0x000) / 4, + // LEDC_TIMERn_VALUE: the live counters. + (0x0a4 - 0x000) / 4, (0x0ac - 0x000) / 4, (0x0b4 - 0x000) / 4, (0x0bc - 0x000) / 4, + // LEDC_INT_RAW and LEDC_INT_ST: overflow and fade-end bits latch while the timers run. + (0x0c0 - 0x000) / 4, (0x0c4 - 0x000) / 4, + // LEDC_TIMERn_CNT_CAP: captured counter values. + (0x150 - 0x000) / 4, (0x154 - 0x000) / 4, (0x158 - 0x000) / 4, (0x15c - 0x000) / 4, + }, + // REG_LEDC_APB_CLK_EN, bit 0 of SOC_CLK_CTRL3 (ledc_ll.h:135). Its reset value is 0, so this + // check is not a formality for LEDC: it is the difference between a snapshot and a memory of + // one. + .clock = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL3_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_S), + }, + // The peripheral reset, REG_RST_EN_LEDC, bit 29 of HP_RST_EN1 (ledc_ll.h:150, + // hp_sys_clkrst_reg.h:3497-3503). Sound here where a configure-restore would not be: this + // block has write-to-trigger fields (both PARA_UPs, OVF_CNT_RESET) whose reset value is only + // defined by the reset, and `LEDC_TIMERn_RST` is one of the fields whose reset value is 1 - + // so "write zeros everywhere" would not be a restore at all. Measured safe on this board: + // pulsing it for 1 ms left the console untouched and returned LEDC_CH0_CONF0 to 0. + .restore = .{ .reset_bit = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_HP_RST_EN1_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_S), + } }, + }, + .setup = setup, + .cases = &.{ + // Timer: the whole sequence, at four target frequencies across three duty resolutions. Each + // side computes its own divider - IDF's `ledc_calculate_divisor`, ours `hal.ledc.divisor` - + // so a mismatch in the fixed-point arithmetic lands in LEDC_TIMERn_CONF[22:5] and is caught + // here rather than being argued about. The four dividers are 1250, 500, 2000 and 2083. + .{ .name = "configure_timer_1kHz_13bit", .arg = 1_000, .idf = idfTimer1k13, .ours = ourTimer1k13 }, + .{ .name = "configure_timer_20kHz_10bit", .arg = 20_000, .idf = idfTimer20k10, .ours = ourTimer20k10 }, + .{ .name = "configure_timer_5kHz_10bit", .arg = 5_000, .idf = idfTimer5k10, .ours = ourTimer5k10 }, + .{ .name = "configure_timer_300Hz_14bit", .arg = 300, .idf = idfTimer300_14, .ours = ourTimer300_14 }, + // The divider store and the arithmetic behind it, without the resolution/resume/reset tail. + .{ .name = "clock_divider_only", .arg = 1_250, .idf = idfDivider, .ours = ourDivider }, + .{ .name = "duty_resolution_only", .arg = 13, .idf = idfResolution, .ours = ourResolution }, + .{ .name = "timer_pause", .idf = idfPause, .ours = ourPause }, + .{ .name = "timer_resume", .idf = idfResume, .ours = ourResume }, + .{ .name = "timer_reset", .idf = idfTimerReset, .ours = ourTimerReset }, + // Channel. + .{ .name = "bind_timer", .idf = idfBind, .ours = ourBind }, + .{ .name = "set_hpoint", .arg = 0x400, .idf = idfHpoint, .ours = ourHpoint }, + .{ .name = "set_duty", .arg = 0x1000, .idf = idfDuty4096, .ours = ourDuty4096 }, + .{ .name = "set_duty", .arg = 0, .idf = idfDuty0, .ours = ourDuty0 }, + // Staged and left uncommitted, on both sides. Passes for the same reason the committed cases + // pass, which is the point: the commit is not in the picture the harness takes. + .{ .name = "stage_without_commit", .arg = 0x555, .idf = idfStageOnly, .ours = ourStageOnly }, + .{ .name = "channel_start", .idf = idfStart, .ours = ourStart }, + .{ .name = "channel_stop_idle_low", .arg = 0, .idf = idfStopLow, .ours = ourStopLow }, + .{ .name = "channel_stop_idle_high", .arg = 1, .idf = idfStopHigh, .ours = ourStopHigh }, + .{ .name = "configure_channel", .arg = 0x800, .idf = idfConfigureChannel, .ours = ourConfigureChannel }, + .{ .name = "full_rf_config_25MHz_1bit", .arg = 25, .idf = idfFullRf, .ours = ourFullRf }, + }, +}; + +// -------------------------------------------------------------------------- the gamma RAM window + +/// Zero the whole gamma RAM aperture and pulse the peripheral reset. +/// +/// The zeroing is the load-bearing half. Gamma RAM is RAM: the peripheral reset does *not* clear it, +/// so without this the second run would inherit whatever the first run wrote, and an implementation +/// that wrote no gamma entry at all would compare equal to one that did - the self-consistent test +/// that proves nothing. All 128 words rather than the channel under test's 16, so that the state the +/// two runs start from does not depend on which cases ran before. +fn restoreGamma() void { + var w: u32 = 0; + while (w < 128) : (w += 1) { + mmio.Reg.atAddress(@as(u32, @intCast(regs.LEDC_CH0_GAMMA_RANGE0_REG)) + 4 * w).writeRaw(0); + } + hal.clkrst.resetPeripheral(.ledc); +} + +/// The gamma RAM aperture, 0x400-0x5ff: sixteen entries for each of the eight channels. +/// +/// It has its own suite because it is not contiguous with the register block - between them lies a +/// 0x288-byte hole that nothing documents, and reading unmapped peripheral space to get from one to +/// the other is not a risk worth taking on the only board. +/// +/// What it covers: on the P4 a constant duty is a degenerate one-step fade, because +/// `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` moved out of `LEDC_CHn_CONF1_REG` into this RAM. +/// `setDuty` writes entry 0 accordingly (ledc.c:263-280), and this is the window that sees it. +pub const gamma_suite: types.Suite = .{ + .descriptor = .{ + .name = "ledc_gamma", + .base = @intCast(regs.LEDC_CH0_GAMMA_RANGE0_REG), + .words = 128, + .clock = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL3_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_S), + }, + .restore = .{ .configure = restoreGamma }, + }, + .setup = setup, + .cases = &.{ + .{ .name = "set_duty_writes_entry0", .arg = 0x1000, .idf = idfDuty4096, .ours = ourDuty4096 }, + .{ .name = "set_duty_writes_entry0", .arg = 0, .idf = idfDuty0, .ours = ourDuty0 }, + .{ .name = "configure_channel_writes_entry0", .arg = 0x800, .idf = idfConfigureChannel, .ours = ourConfigureChannel }, + }, +}; + +// ------------------------------------------------------------------------------- the GPIO window + +/// The pad back to a known state: driver off, IO MUX word zeroed, matrix pointing at plain GPIO. +/// The same restore GPIO's own suite uses, for the same reason - there is no reset bit for GPIO and +/// the pads are the board's wiring. +fn restorePad() void { + hal.gpio.outputDisable(pin); + mmio.Reg.atAddress(@as(u32, @intCast(regs.PERIPHS_IO_MUX_U_PAD_GPIO0)) + 4 * @as(u32, pin)).writeRaw(0); + mmio.Reg.atAddress(@as(u32, @intCast(regs.GPIO_FUNC0_OUT_SEL_CFG_REG)) + 4 * @as(u32, pin)) + .writeRaw(hal.gpio.matrix_gpio_signal); + hal.gpio.setLow(pin); +} + +/// Pin routing touches no LEDC register: the peripheral has no pad of its own, and `attachPin` is +/// entirely a GPIO matrix operation. So it is compared in the GPIO window, where its effect is - and +/// what is actually under test here is the signal index, `LEDC_LS_SIG_OUT_PAD_OUT0_IDX + channel`, +/// which is the one piece of arithmetic in the routing path. +pub const routing_suite: types.Suite = .{ + .descriptor = .{ + .name = "ledc_pin", + .base = @intCast(regs.GPIO_OUT_REG - 4), // GPIO_BT_SELECT_REG sits at +0x00 + .words = 400, + .volatile_words = &.{ + (0x03c - 0x000) / 4, // GPIO_IN - the outside world, which moves + (0x040 - 0x000) / 4, // GPIO_IN1 + }, + .restore = .{ .configure = restorePad }, + }, + .cases = &.{ + .{ .name = "attach_pin", .idf = idfAttachPin, .ours = ourAttachPin }, + .{ .name = "attach_pin_channel7", .arg = 7, .idf = idfAttachPin7, .ours = ourAttachPin7 }, + }, +}; + +// -------------------------------------------------------------------------- the HP_SYS_CLKRST word + +/// LEDC's clock mux and function-clock gate back to what `setup` establishes. Only LEDC's own fields +/// are written: PERI_CLK_CTRL22 also holds RMT's, and this is a live board. + +/// Restored through ESP-IDF's side, never through the code under test. `differ.zig` runs restore, +/// idf, snapshot, restore, ours, snapshot: with the HAL on both the restore and the "ours" side, a +/// HAL function that does nothing leaves run B's snapshot equal to run A's and the case passes. That +/// makes a suite blind to precisely the failure it was written to catch. +fn restoreClk() void { + oracle_ledc_set_clock_source(0); // 0 = XTAL, the value idfSrcXtal uses + oracle_ledc_enable_function_clock(1); +} + +/// One word: `HP_SYS_CLKRST_PERI_CLK_CTRL22_REG`, which on the P4 holds LEDC's clock source select +/// and its function-clock gate (ledc_ll.h:179, :241). This is where the LEDC clock source lives on +/// this die - not in `LEDC_CONF_REG.APB_CLK_SEL`, which the register map still documents with a +/// *different* encoding and which IDF's P4 LL never writes. A HAL that wrote the in-block register +/// would pass every case in the `ledc` suite above and produce no PWM at all; this window is what +/// makes that visible. +/// +/// The case order matters: the last case must leave the function clock on and the source at XTAL, +/// because the harness restores *before* each case and not after the last one. +pub const clock_suite: types.Suite = .{ + .descriptor = .{ + .name = "ledc_clk", + .base = @intCast(regs.HP_SYS_CLKRST_PERI_CLK_CTRL22_REG), + .words = 1, + .restore = .{ .configure = restoreClk }, + }, + .setup = setup, + .cases = &.{ + .{ .name = "clock_source_rc_fast", .arg = 1, .idf = idfSrcRcFast, .ours = ourSrcRcFast }, + .{ .name = "clock_source_pll_div", .arg = 2, .idf = idfSrcPllDiv, .ours = ourSrcPllDiv }, + .{ .name = "clock_source_xtal", .arg = 0, .idf = idfSrcXtal, .ours = ourSrcXtal }, + .{ .name = "function_clock_off", .arg = 0, .idf = idfFuncClkOff, .ours = ourFuncClkOff }, + .{ .name = "function_clock_on", .arg = 1, .idf = idfFuncClkOn, .ours = ourFuncClkOn }, + }, +}; + +/// All four windows, in the order they should run: the block first, because a failure there explains +/// failures in the other three. +pub const suites = [_]types.Suite{ suite, gamma_suite, routing_suite, clock_suite }; + +// --------------------------------------------------------------------------------- the case pairs +// +// `catch {}` rather than `catch unreachable` on the `configureTimer` calls: all four divider values +// are inside the field's range (checked on the host against IDF's own expression), so the error path +// is dead - but if this HAL's validity check ever disagreed with IDF's, doing nothing leaves the +// timer unconfigured and the harness reports a diff, where `unreachable` would be undefined +// behaviour in a ReleaseSmall build and would report nothing. + +fn idfTimer1k13() void { + oracle_ledc_configure_timer(timer, src_hz, 1_000, 13); +} +fn ourTimer1k13() void { + ledc.configureTimer(timer, .{ .src_hz = src_hz, .freq_hz = 1_000, .resolution = 13 }) catch {}; +} +fn idfTimer20k10() void { + oracle_ledc_configure_timer(timer, src_hz, 20_000, 10); +} +fn ourTimer20k10() void { + ledc.configureTimer(timer, .{ .src_hz = src_hz, .freq_hz = 20_000, .resolution = 10 }) catch {}; +} +fn idfTimer5k10() void { + oracle_ledc_configure_timer(timer, src_hz, 5_000, 10); +} +fn ourTimer5k10() void { + ledc.configureTimer(timer, .{ .src_hz = src_hz, .freq_hz = 5_000, .resolution = 10 }) catch {}; +} +fn idfTimer300_14() void { + oracle_ledc_configure_timer(timer, src_hz, 300, 14); +} +fn ourTimer300_14() void { + ledc.configureTimer(timer, .{ .src_hz = src_hz, .freq_hz = 300, .resolution = 14 }) catch {}; +} + +// Each side computes the divider with its own arithmetic and stores it with its own code: 40 MHz, +// 1 kHz, 13 bits, which is 1250 = 0x4E2 = 4.8828 in Q10.8. +fn idfDivider() void { + oracle_ledc_set_clock_divider(timer, oracle_ledc_divisor(src_hz, 1_000, 1 << 13)); + oracle_ledc_commit_timer(timer); +} +fn ourDivider() void { + ledc.setClockDivider(timer, ledc.divisor(src_hz, 1_000, 13)); + ledc.commitTimer(timer); +} + +fn idfResolution() void { + oracle_ledc_set_duty_resolution(timer, 13); + oracle_ledc_commit_timer(timer); +} +fn ourResolution() void { + ledc.setDutyResolution(timer, 13); + ledc.commitTimer(timer); +} + +fn idfPause() void { + oracle_ledc_pause_timer(timer); +} +fn ourPause() void { + ledc.pauseTimer(timer); +} +fn idfResume() void { + oracle_ledc_resume_timer(timer); +} +fn ourResume() void { + ledc.resumeTimer(timer); +} +fn idfTimerReset() void { + oracle_ledc_reset_timer(timer); +} +fn ourTimerReset() void { + ledc.resetTimer(timer); +} + +fn idfBind() void { + oracle_ledc_bind_timer(channel, timer); + oracle_ledc_commit_channel(channel); +} +fn ourBind() void { + ledc.bindTimer(channel, timer); + ledc.commitChannel(channel); +} + +fn idfHpoint() void { + oracle_ledc_set_hpoint(channel, 0x400); + oracle_ledc_commit_channel(channel); +} +fn ourHpoint() void { + ledc.setHpoint(channel, 0x400); + ledc.commitChannel(channel); +} + +fn idfDuty4096() void { + oracle_ledc_set_duty(channel, 0x1000); + oracle_ledc_commit_channel(channel); +} +fn ourDuty4096() void { + ledc.setDuty(channel, 0x1000); + ledc.commitChannel(channel); +} +fn idfDuty0() void { + oracle_ledc_set_duty(channel, 0); + oracle_ledc_commit_channel(channel); +} +fn ourDuty0() void { + ledc.setDuty(channel, 0); + ledc.commitChannel(channel); +} + +// No commit on either side. The staged duty and gamma entry must still match. +fn idfStageOnly() void { + oracle_ledc_set_duty(channel, 0x555); +} +fn ourStageOnly() void { + ledc.setDuty(channel, 0x555); +} + +fn idfStart() void { + oracle_ledc_start(channel); +} +fn ourStart() void { + ledc.start(channel); +} +fn idfStopLow() void { + oracle_ledc_stop(channel, 0); +} +fn ourStopLow() void { + ledc.stop(channel, 0); +} +fn idfStopHigh() void { + oracle_ledc_stop(channel, 1); +} +fn ourStopHigh() void { + ledc.stop(channel, 1); +} + +fn idfConfigureChannel() void { + oracle_ledc_configure_channel(channel, timer, 0x800, 0x200, 1, 1); +} +fn ourConfigureChannel() void { + ledc.configureChannel(channel, .{ + .timer = timer, + .duty = 0x800, + .hpoint = 0x200, + .idle_level = 1, + .output_enabled = true, + }); +} + +fn idfAttachPin() void { + oracle_ledc_set_pin(pin, channel); +} +fn ourAttachPin() void { + ledc.attachPin(channel, pin); +} +// Channel 7 explicitly, because the signal index is arithmetic on the channel number and 0 is the +// one value that cannot catch an off-by-one in it. +fn idfAttachPin7() void { + oracle_ledc_set_pin(pin, 7); +} +fn ourAttachPin7() void { + ledc.attachPin(7, pin); +} + +/// The report's RF configuration, end to end: 1-bit resolution at 25 MHz, duty 1, hpoint 0, on +/// channel 0 / timer 0. Reproduced from the ESP-IDF firmware in 02-esp32p4-m3-radio/main/main.c:77-94. +/// +/// This case exists because the two implementations disagree *on the die* for exactly this +/// configuration and nothing smaller: the IDF firmware's carrier toggles GPIO20 at 25 MHz (proven by +/// its own ADC witness catching both rails), and this project's HAL leaves the pad static, while +/// every individual register operation compares equal. So the difference is in the composition, and +/// comparing the whole block after each full bring-up is the only thing that can localise it. +/// Note the source: 80 MHz, not this suite's default `src_hz` (which is XTAL at 40 MHz). At 40 MHz a +/// 1-bit 25 MHz target needs divider 205, below the legal minimum of 256, and the two sides then +/// disagree for a reason that has nothing to do with the RF experiment: this HAL rejects it with +/// DividerOutOfRange while ESP-IDF's *LL* programs it anyway, because IDF's range check lives one +/// layer up in ledc.c rather than in the LL. Worth knowing - it means an IDF LL caller can silently +/// program an illegal divider - but it is not what this case is for. +fn idfFullRf() void { + oracle_ledc_configure_timer(0, ledc.pll_div_hz, 25_000_000, 1); + oracle_ledc_configure_channel(0, 0, 1, 0, 0, 1); +} +fn ourFullRf() void { + ledc.configureTimer(0, .{ .src_hz = ledc.pll_div_hz, .freq_hz = 25_000_000, .resolution = 1 }) catch return; + ledc.configureChannel(0, .{ .timer = 0, .duty = 1, .hpoint = 0, .idle_level = 0 }); +} + +fn idfSrcXtal() void { + oracle_ledc_set_clock_source(0); +} +fn ourSrcXtal() void { + ledc.setClockSource(.xtal); +} +fn idfSrcRcFast() void { + oracle_ledc_set_clock_source(1); +} +fn ourSrcRcFast() void { + ledc.setClockSource(.rc_fast); +} +fn idfSrcPllDiv() void { + oracle_ledc_set_clock_source(2); +} +fn ourSrcPllDiv() void { + ledc.setClockSource(.pll_div); +} +fn idfFuncClkOff() void { + oracle_ledc_enable_function_clock(0); +} +fn ourFuncClkOff() void { + ledc.setFunctionClockEnabled(false); +} +fn idfFuncClkOn() void { + oracle_ledc_enable_function_clock(1); +} +fn ourFuncClkOn() void { + ledc.setFunctionClockEnabled(true); +} + diff --git a/src/oracle/ledc_ref.c b/src/oracle/ledc_ref.c new file mode 100644 index 0000000..6b1403a --- /dev/null +++ b/src/oracle/ledc_ref.c @@ -0,0 +1,210 @@ +/* LEDC's reference implementation: ESP-IDF's own LL, given external linkage. + * + * There is no logic here except where a comment says otherwise, and there is exactly one such + * place - `oracle_ledc_divisor` - because the divider arithmetic lives in a `static inline` inside + * `esp_driver_ledc/src/ledc.c` and is therefore unreachable from a header. It is transcribed + * character for character, with the line number, so that the on-die comparison covers the + * arithmetic and not only the store that follows it. + */ + +/* IDF's clock and reset LL functions are shadowed by a wrapper macro that references + * `__DECLARE_RCC_ATOMIC_ENV`, an identifier IDF never defines anywhere; its purpose is to make an + * unguarded call fail to compile, because the only legal caller holds a spinlock. There is no + * FreeRTOS here and core 1 is held in reset at power-on, so declaring the name is exactly as safe + * as the spinlock would be - and it is what IDF's own bootloader does. */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); + +/* `ledc_ll_set_slow_clk_sel` and `ledc_ll_get_slow_clk_sel` call `abort()` in the default arm of + * their switch (ledc_ll.h:238, :273). Freestanding, nothing declares it; the arms below are all + * reached with constants, so the call folds away and no definition is needed. */ +void abort(void); + +#include <stdint.h> + +#include "hal/ledc_ll.h" +#include "hal/gpio_ll.h" +#include "soc/gpio_struct.h" +#include "soc/gpio_sig_map.h" + +/* P4 has one speed mode: low. `ledc_ll.h` still takes the parameter because the LL is shared with + * parts that have two. */ +#define MODE LEDC_LOW_SPEED_MODE + +/* ---------------------------------------------------------------- clocks, outside the LEDC block */ + +void oracle_ledc_enable_bus_clock(int enable) +{ + ledc_ll_enable_bus_clock(enable != 0); +} + +void oracle_ledc_enable_function_clock(int enable) +{ + ledc_ll_enable_clock(LEDC_LL_GET_HW(), enable != 0); +} + +/* `sel` is this project's `ClockSource` enum, which is HP_SYS_CLKRST's own encoding: 0 XTAL, + * 1 RC_FAST, 2 PLL_DIV. Split into three constant calls so that IDF's switch folds and its + * `abort()` arm never reaches the linker. */ +void oracle_ledc_set_clock_source(unsigned sel) +{ + switch (sel) { + case 0: + ledc_ll_set_slow_clk_sel(LEDC_LL_GET_HW(), LEDC_SLOW_CLK_XTAL); + break; + case 1: + ledc_ll_set_slow_clk_sel(LEDC_LL_GET_HW(), LEDC_SLOW_CLK_RC_FAST); + break; + case 2: + ledc_ll_set_slow_clk_sel(LEDC_LL_GET_HW(), LEDC_SLOW_CLK_PLL_DIV); + break; + default: + break; + } +} + +/* ---------------------------------------------------------------------------- divider arithmetic */ + +/* Verbatim from esp_driver_ledc/src/ledc.c:468-497 (v6.0.2), which is `static inline` in a .c file + * and so cannot be called. The 32-bit wrap of `freq_hz * precision` and the truncation of the + * 64-bit quotient into `uint32_t` are IDF's, and are the whole reason this exists: they are what + * src/hal/ledc.zig's `divisor` has to reproduce. */ +uint32_t oracle_ledc_divisor(uint32_t src_clk_freq, int freq_hz, uint32_t precision) +{ + return (((uint64_t) src_clk_freq << LEDC_LL_FRACTIONAL_BITS) + freq_hz * precision / 2) + / (freq_hz * precision); +} + +/* ------------------------------------------------------------------------------------- timers */ + +void oracle_ledc_set_clock_divider(unsigned timer, uint32_t div) +{ + ledc_ll_set_clock_divider(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer, div); +} + +void oracle_ledc_set_duty_resolution(unsigned timer, uint32_t bits) +{ + ledc_ll_set_duty_resolution(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer, bits); +} + +void oracle_ledc_commit_timer(unsigned timer) +{ + ledc_ll_ls_timer_update(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); +} + +void oracle_ledc_reset_timer(unsigned timer) +{ + ledc_ll_timer_rst(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); +} + +void oracle_ledc_pause_timer(unsigned timer) +{ + ledc_ll_timer_pause(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); +} + +void oracle_ledc_resume_timer(unsigned timer) +{ + ledc_ll_timer_resume(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); +} + +/* `ledc_set_timer_params` (ledc.c:244-261) followed by the resume/reset pair `ledc_timer_config` + * does on success (ledc.c:816-818). The clock-source step of `ledc_set_timer_params` is absent on + * purpose: on the P4 there is no timer-specific mux (SOC_LEDC_HAS_TIMER_SPECIFIC_MUX is unset), so + * that step compiles out of IDF too. */ +void oracle_ledc_configure_timer(unsigned timer, uint32_t src_hz, int freq_hz, uint32_t resolution) +{ + uint32_t div = oracle_ledc_divisor(src_hz, freq_hz, 1u << resolution); + ledc_ll_set_clock_divider(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer, div); + ledc_ll_set_duty_resolution(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer, resolution); + ledc_ll_ls_timer_update(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); + ledc_ll_timer_resume(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); + ledc_ll_timer_rst(LEDC_LL_GET_HW(), MODE, (ledc_timer_t)timer); +} + +/* ------------------------------------------------------------------------------------ channels */ + +void oracle_ledc_bind_timer(unsigned channel, unsigned timer) +{ + ledc_ll_bind_channel_timer(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, (ledc_timer_t)timer); +} + +void oracle_ledc_set_hpoint(unsigned channel, uint32_t hpoint) +{ + ledc_ll_set_hpoint(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, hpoint); +} + +/* `ledc_duty_config` (ledc.c:263-280) with `hpoint_val` left alone: the duty integer part, then the + * degenerate one-step fade in gamma RAM entry 0 that a constant duty needs on this die, then the + * range count. `ledc_hal_clear_left_off_fade_param` is deliberately not called - it zeroes ranges + * 1..15, which only matters once real fades are in scope. */ +void oracle_ledc_set_duty(unsigned channel, uint32_t duty) +{ + ledc_ll_set_duty_int_part(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, duty); + ledc_ll_set_fade_param_range(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, 0, 1, 1, 0, 1); + ledc_ll_set_range_number(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, 1); +} + +void oracle_ledc_set_output_enabled(unsigned channel, int enable) +{ + ledc_ll_set_sig_out_en(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, enable != 0); +} + +void oracle_ledc_set_idle_level(unsigned channel, uint32_t level) +{ + ledc_ll_set_idle_level(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, level); +} + +void oracle_ledc_start_fade(unsigned channel) +{ + ledc_ll_set_duty_start(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); +} + +void oracle_ledc_commit_channel(unsigned channel) +{ + ledc_ll_ls_channel_update(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); +} + +/* `_ledc_update_duty`, ledc.c:1021-1026. */ +void oracle_ledc_start(unsigned channel) +{ + ledc_ll_set_sig_out_en(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, true); + ledc_ll_set_duty_start(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); + ledc_ll_ls_channel_update(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); +} + +/* `ledc_stop`, ledc.c:1039-1050: idle level staged before the output is disabled, one commit. */ +void oracle_ledc_stop(unsigned channel, uint32_t idle_level) +{ + ledc_ll_set_idle_level(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, idle_level); + ledc_ll_set_sig_out_en(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, false); + ledc_ll_ls_channel_update(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); +} + +/* The register half of `ledc_channel_config` (ledc.c:869-1019): stage timer, hpoint, duty, idle + * level and output enable, hand the duty over, commit once. */ +void oracle_ledc_configure_channel(unsigned channel, unsigned timer, uint32_t duty, uint32_t hpoint, + uint32_t idle_level, int output_enabled) +{ + ledc_ll_bind_channel_timer(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, (ledc_timer_t)timer); + ledc_ll_set_hpoint(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, hpoint); + ledc_ll_set_duty_int_part(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, duty); + ledc_ll_set_fade_param_range(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, 0, 1, 1, 0, 1); + ledc_ll_set_range_number(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, 1); + ledc_ll_set_idle_level(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, idle_level); + ledc_ll_set_sig_out_en(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel, output_enabled != 0); + ledc_ll_set_duty_start(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); + ledc_ll_ls_channel_update(LEDC_LL_GET_HW(), MODE, (ledc_channel_t)channel); +} + +/* ---------------------------------------------------------------------------------- pin routing */ + +/* The hardware effect of `ledc_set_pin` (ledc.c:823-836). `gpio_matrix_output` is + * `gpio_hal_matrix_out` (gpio_hal.c:60-69): pad function, matrix source, then the output-enable + * control last "to avoid undesired level change". The signal index is + * `ledc_periph_signal[0].sig_out0_idx + channel`, and that field is initialised to + * `LEDC_LS_SIG_OUT_PAD_OUT0_IDX` in esp_hal_ledc/esp32p4/ledc_periph.c:14-18. */ +void oracle_ledc_set_pin(unsigned pin, unsigned channel) +{ + gpio_ll_func_sel(&GPIO, pin, PIN_FUNC_GPIO); + gpio_ll_set_output_signal_matrix_source(&GPIO, pin, LEDC_LS_SIG_OUT_PAD_OUT0_IDX + channel, false); + gpio_ll_set_output_enable_ctrl(&GPIO, pin, true, false); +} diff --git a/src/oracle/oracle_sdkconfig.h b/src/oracle/oracle_sdkconfig.h new file mode 100644 index 0000000..c74fb50 --- /dev/null +++ b/src/oracle/oracle_sdkconfig.h @@ -0,0 +1,43 @@ +/* The Kconfig surface ESP-IDF's LL headers are compiled against when they are used as the + * differential reference. Deliberately minimal and deliberately *ours*. + * + * An earlier attempt borrowed sdkconfig.h from an unrelated ESP-IDF project in this workspace. That + * is a trap with a measurable cost: the borrowed file sets CONFIG_HAL_GPIO_USE_ROM_IMPL=1, which + * makes gpio_ll_set_level() call rom_gpio_set_output_level() and write no GPIO register at all - + * so the very first differential would have compared this HAL against the mask ROM rather than + * against IDF's register sequence. Recompiling the same LL headers against a different sdkconfig + * changes the emitted .text of six of nine tier-1/2 peripherals, so this file is part of the + * experiment's definition, not incidental. + * + * Anything not defined here is simply absent, which for IDF's `#if` tests means zero. That is the + * behaviour we want: the register path, with nothing optional switched on. + */ +#pragma once + +/* Target selection. Everything under components/soc and components/hal keys off this. */ +#define CONFIG_IDF_TARGET_ESP32P4 1 +#define CONFIG_IDF_TARGET "esp32p4" + +/* Pre-v3 silicon: this die is rev v1.3. The same condition selects register/hw_ver1 in IDF's own + * build (soc/CMakeLists.txt:37-41) and esp32p4.rom.ld rather than esp32p4.rom.eco5.ld - 237 of 452 + * common ROM symbols have different addresses between those two files, so the pairing is not + * cosmetic. build.zig asserts the register module was built from hw_ver1 to match. */ +#define CONFIG_ESP32P4_SELECTS_REV_LESS_V3 1 +#define CONFIG_ESP32P4_REV_MIN_FULL 100 +#define CONFIG_ESP32P4_REV_MAX_FULL 199 + +/* 40 MHz crystal, as fitted. Reaches the HAL through HAL_CONFIG_XTAL_HINT_FREQ_MHZ. */ +#define CONFIG_XTAL_FREQ 40 + +/* Assertions off, and this one is a real choice rather than tidiness: at level 2 HAL_ASSERT expands + * to __assert_func (a libc symbol this image does not have), and at 0 it becomes + * __builtin_unreachable(), which lets clang delete the argument-checking branches. The reference + * implementation should be the code IDF ships in a release build, and a harness that wants to test + * argument validation must not rely on a branch the compiler is entitled to remove. */ +#define CONFIG_HAL_DEFAULT_ASSERTION_LEVEL 0 + +/* NOT defined, on purpose: + * CONFIG_HAL_GPIO_USE_ROM_IMPL - would route gpio_ll_set_level through the mask ROM (see above). + * CONFIG_IDF_ENV_FPGA - would change efuse and clock behaviour to the FPGA model. + * CONFIG_PM_*, CONFIG_FREERTOS_* - no power management and no OS in this image. + */ diff --git a/src/oracle/runtime_ref.c b/src/oracle/runtime_ref.c new file mode 100644 index 0000000..659661b --- /dev/null +++ b/src/oracle/runtime_ref.c @@ -0,0 +1,19 @@ +/* The few libc symbols ESP-IDF's LL code reaches for, supplied so the reference can link into a + * freestanding image. + * + * There is exactly one so far, and it is reached by design rather than by accident: + * `_uart_ll_set_baudrate` (uart_ll.h:532-535) calls `abort()` when handed an LP_UART instance, + * because that path needs `lp_uart_ll_set_baudrate` instead. The differential harness only ever + * passes HP UART instances, so this is unreachable in practice - but the linker does not know that, + * and a missing `abort` fails the build with a symbol name that explains nothing about why. + * + * Spinning rather than resetting is deliberate: if a reference implementation ever does call this, + * the board stops with its last console line intact, which is the difference between a diagnosable + * failure and a reboot loop. + */ + +__attribute__((noreturn)) void abort(void) +{ + for (;;) { + } +} diff --git a/src/oracle/sdmmc_cases.zig b/src/oracle/sdmmc_cases.zig new file mode 100644 index 0000000..4b53aed --- /dev/null +++ b/src/oracle/sdmmc_cases.zig @@ -0,0 +1,567 @@ +//! SDMMC's side of the differential test. +//! +//! Two windows, because this peripheral's state is not contiguous. The controller's own register +//! block is at 0x50083000; its *host* clock generator - source mux, two-stage divider, sampling +//! phase - is not in it at all, but in HP_SYS_CLKRST, where the P4 moved it. That is the same +//! split I2C has (`i2c.clock_suite`), and for the same reason: a driver that programmed every +//! register inside the block perfectly and the divider not at all would run the bus at the wrong +//! frequency and pass every case in the first suite. +//! +//! **No case sends a command to the card.** The five `cmd_word_*` cases write the command register +//! with `start_command` (bit 31) cleared, which is what makes them safe: bit 31 is the launch, and +//! a word without it is inert. The C6 is in reset for the whole of a differ run - GPIO54 is never +//! released - so a real CMD52 would sit out its response timeout and prove nothing. What is being +//! compared is the encoding, and the encoding is entirely visible in the staged word. +//! +//! **Restore is the peripheral's own reset**, LP_AON_CLKRST bit 28, which is legitimate here and +//! not merely convenient: this block is full of self-clearing and write-1-to-clear bits (the three +//! reset bits in CTRL, every bit of RINTSTS, the IDMAC's software reset), and writing a snapshot +//! back would trigger a reset rather than undo one. Nothing in this file restores through the code +//! under test; the only Zig the harness runs between the two halves is `mmio`. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +extern fn oracle_sdmmc_bus_clock(enable: c_int) void; +extern fn oracle_sdmmc_reset_register() void; +extern fn oracle_sdmmc_set_host_clock_div(div: c_uint) void; +extern fn oracle_sdmmc_select_clk_source_pll160m() void; +extern fn oracle_sdmmc_init_phase_delay() void; +extern fn oracle_sdmmc_set_card_clock_div(slot: c_uint, div: c_uint) void; +extern fn oracle_sdmmc_enable_card_clock(slot: c_uint, enable: c_int) void; +extern fn oracle_sdmmc_enable_card_clock_low_power(slot: c_uint, enable: c_int) void; +extern fn oracle_sdmmc_reset_controller() void; +extern fn oracle_sdmmc_reset_dma() void; +extern fn oracle_sdmmc_reset_fifo() void; +extern fn oracle_sdmmc_module_reset() void; +extern fn oracle_sdmmc_set_card_width(slot: c_uint, width: c_uint) void; +extern fn oracle_sdmmc_set_block_size(size: c_uint) void; +extern fn oracle_sdmmc_set_data_transfer_len(len: c_uint) void; +extern fn oracle_sdmmc_set_timeouts(data_cycles: c_uint, response_cycles: c_uint) void; +extern fn oracle_sdmmc_set_fifo_threshold(rx: c_uint, tx: c_uint, msize: c_uint) void; +extern fn oracle_sdmmc_configure_interrupts() void; +extern fn oracle_sdmmc_init_dma() void; +extern fn oracle_sdmmc_enable_dma(enable: c_int) void; +extern fn oracle_sdmmc_set_desc_addr(addr: c_uint) void; +extern fn oracle_sdmmc_enable_sdio_interrupt(slot: c_uint, enable: c_int) void; +extern fn oracle_sdmmc_stage_command( + index: c_uint, + response_long: c_int, + response_expect: c_int, + check_crc: c_int, + data: c_int, + send_init: c_int, + wait_prvdata: c_int, + update_clk: c_int, + slot: c_uint, +) void; +extern fn oracle_sdmmc_version_id() c_uint; +extern fn oracle_sdmmc_hw_config() c_uint; + +/// Printed by the harness's caller, so a run records which controller it was talking to. A version +/// ID of 0 or 0xffffffff means the block is gated or absent and every result below is noise. +pub fn versionId() u32 { + return oracle_sdmmc_version_id(); +} + +pub fn hwConfig() u32 { + return oracle_sdmmc_hw_config(); +} + +// Force the whole of `hal/sdmmc.zig` through the compiler for the *chip*. +// +// Zig analyses a function only when something references it, and this is the only build that +// compiles that file for riscv32 at all - the plain application never mentions SDMMC, and the +// host test root reaches only the pure encoding functions (it cannot reach the rest: reading the +// `cycle` CSR does not assemble for x86). So without this list, `cmd53Read`, `cardInit` and the +// whole transfer path would be text that has never been type-checked against the target, which is +// a bad thing to discover on a board. +// +// A `-Doracle` build failing here is the intended behaviour: it means the driver does not +// compile, and it says so before anything is flashed. +comptime { + _ = &hal.sdmmc.init; + _ = &hal.sdmmc.cardInit; + _ = &hal.sdmmc.cmd52Read; + _ = &hal.sdmmc.cmd52Write; + _ = &hal.sdmmc.cmd53Read; + _ = &hal.sdmmc.cmd53Write; + _ = &hal.sdmmc.slaveInterruptPending; + _ = &hal.sdmmc.clearSlaveInterrupt; + _ = &hal.sdmmc.setSlaveInterruptEnabled; + _ = &hal.sdmmc.rca; + _ = &hal.sdmmc.configurePins; + _ = &hal.sdmmc.setBusClock; + _ = &hal.sdmmc.dividersFor; + _ = &hal.sdmmc.cmd52Arg; + _ = &hal.sdmmc.cmd53Arg; + _ = hal.sdmmc.interrupt_source; + _ = hal.sdmmc.bounce_len; + _ = hal.sdmmc.c6_pins; +} + +/// The slot under test. Slot 1 is where the ESP32-C6 is; slot 0's pads are the P4's own flash +/// interface on this board and are never touched. +const slot: u1 = 1; + +const cmd_reg = mmio.Reg.atAddress(@intCast(regs.SDHOST_CMD_REG)); + +/// Stage the word our HAL would send, with the launch bit removed. `hal.sdmmc.commandWord` is the +/// code under test; the store is one line and is not. +fn stage(c: hal.sdmmc.Command) void { + cmd_reg.writeRaw(hal.sdmmc.commandWord(c) & ~(@as(u32, 1) << 31)); +} + +// -------------------------------------------------------------------------- the register block + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "sdmmc", + .base = @intCast(regs.SDHOST_CTRL_REG), // offset 0 of the block + // 0x000 through ENSHIFT at +0x110. The window deliberately stops short of BUFFIFO at + // +0x200: that is the data FIFO, and a snapshot loop that read it would pop received + // words - the same hazard `UART_FIFO_REG` poses at offset 0 of every UART. The three + // registers above it (CLK_EDGE_SEL, RAW_INTS, DLL_CLK_CONF at +0x800) belong to the + // high-speed delay-line path this driver does not use. + .words = 69, + .volatile_words = &.{ + (0x40 - 0x00) / 4, // MINTSTS - the C6 can raise its SDIO interrupt at any moment + (0x44 - 0x00) / 4, // RINTSTS - likewise, and write-1-to-clear + (0x48 - 0x00) / 4, // STATUS - FIFO count, FSM state, live DAT levels + (0x50 - 0x00) / 4, // CDETECT - a live input + (0x54 - 0x00) / 4, // WRTPRT - a live input + (0x5c - 0x00) / 4, // TCBCNT - transferred card byte count + (0x60 - 0x00) / 4, // TBBCNT - transferred host byte count + (0x8c - 0x00) / 4, // IDSTS - IDMAC status, write-1-to-clear + (0x94 - 0x00) / 4, // DSCADDR - the IDMAC's current descriptor pointer + (0x98 - 0x00) / 4, // BUFADDR - the IDMAC's current buffer pointer + }, + // Unlike most of this chip, SDMMC powers up with its bus clock *off* + // (HP_SYS_CLKRST SOC_CLK_CTRL1 REG_SDMMC_SYS_CLK_EN, default 0), so this check is the one + // that catches a setup that silently did not happen: a gated block returns the last value + // latched, not zeros, and two such snapshots compare equal while describing nothing. + .clock = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_S), + }, + // LP_AON_CLKRST.hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc - `sdmmc_ll.h:158-163`. Not in + // HP_SYS_CLKRST with almost every other peripheral's reset, which is the single most + // surprising fact about this block's clock and reset wiring. + .restore = .{ .reset_bit = .{ + .reg = @intCast(regs.LP_CLKRST_HP_SDMMC_EMAC_RST_CTRL_REG), + .bit = @intCast(regs.LP_CLKRST_RST_EN_SDMMC_S), + } }, + }, + .cases = &.{ + // --- resets. Each of the three bits is self-clearing, so what these compare is mostly + // that the *other* bits of CTRL come out the same: a reset function that wrote bit 5 + // (dma_enable) instead of bit 2 (dma_reset) would leave a trace, and that is exactly the + // kind of slip the two undocumented CTRL bits invite. + .{ .name = "reset_controller", .idf = idfResetCtl, .ours = ourResetCtl }, + .{ .name = "reset_fifo", .idf = idfResetFifo, .ours = ourResetFifo }, + .{ .name = "reset_dma", .idf = idfResetDma, .ours = ourResetDma }, + .{ .name = "module_reset", .idf = idfModuleReset, .ours = ourModuleReset }, + // --- the card clock: CLKDIV, CLKSRC, CLKENA. Divider 0 is bypass (40 MHz through the + // host divider alone); divider 20 is the 400 kHz probing setting. + .{ .name = "card_clock_div", .arg = 0, .idf = idfCardDiv0, .ours = ourCardDiv0 }, + .{ .name = "card_clock_div", .arg = 20, .idf = idfCardDiv20, .ours = ourCardDiv20 }, + .{ .name = "card_clock_enable", .arg = 1, .idf = idfCclkOn, .ours = ourCclkOn }, + .{ .name = "card_clock_low_power", .arg = 0, .idf = idfLpOff, .ours = ourLpOff }, + .{ .name = "card_clock_low_power", .arg = 1, .idf = idfLpOn, .ours = ourLpOn }, + // --- bus width. The measured working dump has ctype=0x00000002, i.e. bit 1: slot 1 in + // 4-bit mode, which is what `bus_width(4)` must produce and nothing else. + .{ .name = "bus_width", .arg = 4, .idf = idfWidth4, .ours = ourWidth4 }, + .{ .name = "bus_width", .arg = 1, .idf = idfWidth1, .ours = ourWidth1 }, + // --- transfer geometry. + .{ .name = "block_size", .arg = 512, .idf = idfBlk512, .ours = ourBlk512 }, + .{ .name = "block_size", .arg = 4, .idf = idfBlk4, .ours = ourBlk4 }, + .{ .name = "timeouts", .idf = idfTimeouts, .ours = ourTimeouts }, + // A deliberately non-default watermark set, so the case is not "both wrote the reset + // value". ESP-IDF has no LL function for FIFOTH at all and never writes the register on + // any target, so the reference here goes through IDF's `SDMMC.fifoth` bitfields instead - + // which is still IDF's definition of where those three fields sit. + .{ .name = "fifo_threshold", .arg = 255, .idf = idfFifoth, .ours = ourFifoth }, + .{ .name = "fifo_threshold_default", .arg = 511, .idf = idfFifothDefault, .ours = ourFifothDefault }, + // --- interrupts and DMA. + .{ .name = "configure_interrupts", .idf = idfIntrs, .ours = ourIntrs }, + .{ .name = "sdio_interrupt", .arg = 1, .idf = idfSdioIntOn, .ours = ourSdioIntOn }, + .{ .name = "init_dma", .idf = idfInitDma, .ours = ourInitDma }, + .{ .name = "desc_addr", .idf = idfDescAddr, .ours = ourDescAddr }, + // --- command-word encodings. The five commands `cardInit` sends, plus both directions of + // CMD53 and the clock update command that is not a command at all. + .{ .name = "cmd_word_cmd0", .arg = 0, .idf = idfCmd0, .ours = ourCmd0 }, + .{ .name = "cmd_word_cmd5", .arg = 5, .idf = idfCmd5, .ours = ourCmd5 }, + .{ .name = "cmd_word_cmd3", .arg = 3, .idf = idfCmd3, .ours = ourCmd3 }, + .{ .name = "cmd_word_cmd7", .arg = 7, .idf = idfCmd7, .ours = ourCmd7 }, + .{ .name = "cmd_word_cmd52_read", .arg = 52, .idf = idfCmd52R, .ours = ourCmd52R }, + .{ .name = "cmd_word_cmd52_write", .arg = 52, .idf = idfCmd52W, .ours = ourCmd52W }, + .{ .name = "cmd_word_cmd53_read", .arg = 53, .idf = idfCmd53R, .ours = ourCmd53R }, + .{ .name = "cmd_word_cmd53_write", .arg = 53, .idf = idfCmd53W, .ours = ourCmd53W }, + .{ .name = "cmd_word_clock_update", .idf = idfCmdClk, .ours = ourCmdClk }, + // --- the peripheral reset itself, which is in LP_AON_CLKRST and observable here only by + // its effect: configure the block distinctively through IDF's LL on both sides, then let + // each implementation reset it. A `resetPeripheral(.sdmmc)` that wrote the wrong bit - + // there is no HP_SYS_CLKRST reset for SDMMC, so writing one is the obvious mistake - would + // leave the configuration standing. + .{ .name = "peripheral_reset", .idf = idfPeriphReset, .ours = ourPeriphReset }, + }, + .setup = setup, +}; + +/// Bring the block up far enough that its registers are live, and settle the HAL's idea of which +/// slot it is driving. +/// +/// The clock and the reset go through ESP-IDF's LL, not ours: setup runs once, before any case, +/// and a setup written with the code under test would hide a broken `clkrst.init(.sdmmc)` behind +/// its own success. `hal.sdmmc.init` runs afterwards for a different reason - it is the only way +/// to tell the HAL that this is slot 1, and running it here means a bring-up that hangs shows up +/// as a stalled suite rather than as a wrong register somewhere later. Its result is discarded: +/// every case restores the block by resetting it, so nothing init leaves behind is load-bearing, +/// and a card that never answers must not stop the register comparison from running. +fn setup() void { + oracle_sdmmc_bus_clock(1); + oracle_sdmmc_reset_register(); + hal.sdmmc.init(.{ .slot = slot, .width = .four, .khz = 40_000 }) catch {}; +} + +fn idfResetCtl() void { + oracle_sdmmc_reset_controller(); +} +fn ourResetCtl() void { + mmio.Reg.atAddress(@intCast(regs.SDHOST_CTRL_REG)).modify(.{ + mmio.Field.of(regs.SDHOST_CONTROLLER_RESET_S, regs.SDHOST_CONTROLLER_RESET_V).is(1), + }); +} +fn idfResetFifo() void { + oracle_sdmmc_reset_fifo(); +} +fn ourResetFifo() void { + mmio.Reg.atAddress(@intCast(regs.SDHOST_CTRL_REG)).modify(.{ + mmio.Field.of(regs.SDHOST_FIFO_RESET_S, regs.SDHOST_FIFO_RESET_V).is(1), + }); +} +fn idfResetDma() void { + oracle_sdmmc_reset_dma(); +} +fn ourResetDma() void { + mmio.Reg.atAddress(@intCast(regs.SDHOST_CTRL_REG)).modify(.{ + mmio.Field.of(regs.SDHOST_DMA_RESET_S, regs.SDHOST_DMA_RESET_V).is(1), + }); +} +fn idfModuleReset() void { + oracle_sdmmc_module_reset(); +} +fn ourModuleReset() void { + hal.sdmmc.resetController() catch {}; +} + +fn idfCardDiv0() void { + oracle_sdmmc_set_card_clock_div(slot, 0); +} +fn ourCardDiv0() void { + hal.sdmmc.setCardClockDiv(0); +} +fn idfCardDiv20() void { + oracle_sdmmc_set_card_clock_div(slot, 20); +} +fn ourCardDiv20() void { + hal.sdmmc.setCardClockDiv(20); +} + +fn idfCclkOn() void { + oracle_sdmmc_enable_card_clock(slot, 1); +} +fn ourCclkOn() void { + hal.sdmmc.setCardClockEnabled(true); +} +fn idfLpOff() void { + oracle_sdmmc_enable_card_clock_low_power(slot, 0); +} +fn ourLpOff() void { + hal.sdmmc.setCardClockLowPower(false); +} +fn idfLpOn() void { + oracle_sdmmc_enable_card_clock_low_power(slot, 1); +} +fn ourLpOn() void { + hal.sdmmc.setCardClockLowPower(true); +} + +fn idfWidth4() void { + oracle_sdmmc_set_card_width(slot, 4); +} +fn ourWidth4() void { + hal.sdmmc.setBusWidth(.four); +} +fn idfWidth1() void { + oracle_sdmmc_set_card_width(slot, 1); +} +fn ourWidth1() void { + hal.sdmmc.setBusWidth(.one); +} + +fn idfBlk512() void { + oracle_sdmmc_set_block_size(512); + oracle_sdmmc_set_data_transfer_len(512); +} +fn ourBlk512() void { + hal.sdmmc.setBlockSize(512); + hal.sdmmc.setDataTransferLen(512); +} +fn idfBlk4() void { + // The geometry the measured working dump was taken at: blksiz=4 bytcnt=4, the four-byte + // register read ESP-Hosted does to find out how much the slave has queued. + oracle_sdmmc_set_block_size(4); + oracle_sdmmc_set_data_transfer_len(4); +} +fn ourBlk4() void { + hal.sdmmc.setBlockSize(4); + hal.sdmmc.setDataTransferLen(4); +} + +fn idfTimeouts() void { + // 100 ms of card clocks at 40 MHz, and the maximum response timeout - `sd_host_sdmmc.c:531-535`. + oracle_sdmmc_set_timeouts(100 * 40_000, 255); +} +fn ourTimeouts() void { + hal.sdmmc.setTimeouts(100 * 40_000, 255); +} + +fn idfFifoth() void { + oracle_sdmmc_set_fifo_threshold(255, 8, 2); +} +fn ourFifoth() void { + hal.sdmmc.setFifoThreshold(255, 8, 2); +} +fn idfFifothDefault() void { + oracle_sdmmc_set_fifo_threshold(511, 0, 0); +} +fn ourFifothDefault() void { + hal.sdmmc.setFifoThreshold( + hal.sdmmc.default_rx_watermark, + hal.sdmmc.default_tx_watermark, + hal.sdmmc.default_dma_msize, + ); +} + +fn idfIntrs() void { + oracle_sdmmc_configure_interrupts(); +} +fn ourIntrs() void { + hal.sdmmc.configureInterrupts(); +} +fn idfSdioIntOn() void { + oracle_sdmmc_enable_sdio_interrupt(slot, 1); +} +fn ourSdioIntOn() void { + hal.sdmmc.setSlaveInterruptEnabled(true); +} + +fn idfInitDma() void { + oracle_sdmmc_init_dma(); + oracle_sdmmc_enable_dma(1); +} +fn ourInitDma() void { + hal.sdmmc.initDma(); + hal.sdmmc.setDmaEnabled(true); +} + +/// An address in L2MEM with the low bits set to something a bug would round away: DBADDR ignores +/// bits [1:0] internally but stores what is written. +const test_desc_addr: u32 = 0x4ff1_0140; + +fn idfDescAddr() void { + oracle_sdmmc_set_desc_addr(test_desc_addr); +} +fn ourDescAddr() void { + hal.sdmmc.setDescriptorAddr(test_desc_addr); +} + +// The command words. Each pair is the same command expressed twice: once through ESP-IDF's +// `sdmmc_hw_cmd_t` bitfields, once through this project's `commandWord`. + +fn idfCmd0() void { + oracle_sdmmc_stage_command(0, 0, 0, 0, 0, 1, 0, 0, slot); +} +fn ourCmd0() void { + stage(.{ .index = 0, .send_init = true, .wait_prvdata = false, .slot = slot }); +} +fn idfCmd5() void { + oracle_sdmmc_stage_command(5, 0, 1, 0, 0, 0, 1, 0, slot); +} +fn ourCmd5() void { + stage(.{ .index = 5, .response = .short, .check_crc = false, .slot = slot }); +} +fn idfCmd3() void { + oracle_sdmmc_stage_command(3, 0, 1, 1, 0, 0, 1, 0, slot); +} +fn ourCmd3() void { + stage(.{ .index = 3, .response = .short, .check_crc = true, .slot = slot }); +} +fn idfCmd7() void { + oracle_sdmmc_stage_command(7, 0, 1, 1, 0, 0, 1, 0, slot); +} +fn ourCmd7() void { + stage(.{ .index = 7, .response = .short, .check_crc = true, .slot = slot }); +} +fn idfCmd52R() void { + oracle_sdmmc_stage_command(52, 0, 1, 1, 0, 0, 1, 0, slot); +} +fn ourCmd52R() void { + stage(.{ .index = 52, .response = .short, .check_crc = true, .slot = slot }); +} +fn idfCmd52W() void { + // CMD52 carries its payload in the argument, not in a data phase, so the word is identical to + // the read one. Kept as its own case because that is a claim worth checking rather than + // assuming: an implementation that set `rw` for a write would fail here and nowhere else. + oracle_sdmmc_stage_command(52, 0, 1, 1, 0, 0, 1, 0, slot); +} +fn ourCmd52W() void { + stage(.{ .index = 52, .response = .short, .check_crc = true, .slot = slot }); +} +fn idfCmd53R() void { + oracle_sdmmc_stage_command(53, 0, 1, 1, 1, 0, 1, 0, slot); +} +fn ourCmd53R() void { + stage(.{ .index = 53, .response = .short, .check_crc = true, .data = .read, .slot = slot }); +} +fn idfCmd53W() void { + oracle_sdmmc_stage_command(53, 0, 1, 1, 2, 0, 1, 0, slot); +} +fn ourCmd53W() void { + stage(.{ .index = 53, .response = .short, .check_crc = true, .data = .write, .slot = slot }); +} +fn idfCmdClk() void { + oracle_sdmmc_stage_command(0, 0, 0, 0, 0, 0, 1, 1, slot); +} +fn ourCmdClk() void { + stage(.{ .index = 0, .update_clock = true, .slot = slot }); +} + +/// A configuration distinctive enough that failing to clear it is visible in three registers. +fn configureDistinctively() void { + oracle_sdmmc_set_card_width(slot, 4); + oracle_sdmmc_set_block_size(4); + oracle_sdmmc_set_fifo_threshold(255, 8, 2); +} + +fn idfPeriphReset() void { + configureDistinctively(); + oracle_sdmmc_reset_register(); +} +fn ourPeriphReset() void { + configureDistinctively(); + hal.clkrst.resetPeripheral(.sdmmc); +} + +// ------------------------------------------------------------------- the host clock generator + +/// The other half of "set the bus to 40 MHz", which is not in the SDMMC block. +/// +/// `HP_SYS_CLKRST.peri_clk_ctrl01` holds the source mux and the gate, `peri_clk_ctrl02` the +/// three-edge divider and the driving/sampling phase clocks (`sdmmc_ll.h:212-315`). At 40 MHz the +/// host divider is 4 and the card divider is 0, so *all* of the division happens here: an +/// implementation that wrote CLKDIV correctly and this register not at all would clock the C6 at +/// 160 MHz, which is four times the part's limit and would fail as a wiring problem. +pub const clock_suite: types.Suite = .{ + .descriptor = .{ + .name = "sdmmc_clk", + .base = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG - 0x18), // block base + // 0x00 through PERI_CLK_CTRL03 at +0x3c: SOC_CLK_CTRL0..3 (the bus-clock gates) and + // PERI_CLK_CTRL00..03 (the SDIO clock generator). + .words = 16, + // No clock check: HP_SYS_CLKRST is the block that holds every other block's gate and has + // none of its own, and one of the cases below deliberately turns SDMMC's off. + .restore = .{ .configure = restoreClocks }, + }, + .cases = &.{ + // Disable first, so "enable" is not a no-op against a restored state that already has it + // on - the shape clkrst_cases.zig arrived at for the same reason. + .{ .name = "bus_clock", .arg = 0, .idf = idfBusClkOff, .ours = ourBusClkOff }, + .{ .name = "host_clock_div", .arg = 4, .idf = idfHostDiv4, .ours = ourHostDiv4 }, + .{ .name = "host_clock_div", .arg = 8, .idf = idfHostDiv8, .ours = ourHostDiv8 }, + .{ .name = "host_clock_div", .arg = 10, .idf = idfHostDiv10, .ours = ourHostDiv10 }, + .{ .name = "select_clk_source", .idf = idfSelectSrc, .ours = ourSelectSrc }, + .{ .name = "init_phase_delay", .idf = idfPhase, .ours = ourPhase }, + // Last, so the block is left clocked whichever side ran last: every suite after this one + // that touches SDMMC depends on it. + .{ .name = "bus_clock", .arg = 1, .idf = idfBusClkOn, .ours = ourBusClkOn }, + }, +}; + +const soc_clk_ctrl1 = mmio.Reg.atAddress(@intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG)); +const peri01 = mmio.Reg.atAddress(@intCast(regs.HP_SYS_CLKRST_PERI_CLK_CTRL01_REG)); +const peri02 = mmio.Reg.atAddress(@intCast(regs.HP_SYS_CLKRST_PERI_CLK_CTRL02_REG)); + +/// Every SDIO field of the three registers this suite's cases touch, back to its reset value - +/// and nothing else, because these words also hold the gates and clock muxes of peripherals that +/// have nothing to do with SDMMC (MIPI DSI's D-PHY source select is bits 30-31 of PERI_CLK_CTRL02). +/// Built from register macros only; nothing here calls the code under test. +fn restoreClocks() void { + soc_clk_ctrl1.modify(.{ + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_V).is(1), + }); + peri01.modify(.{ + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_S, regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_V).is(0), + }); + peri02.modify(.{ + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_V).is(0), + mmio.Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_V).is(0), + }); +} + +fn idfBusClkOff() void { + oracle_sdmmc_bus_clock(0); +} +fn ourBusClkOff() void { + hal.clkrst.setClockEnabled(.sdmmc, false); +} +fn idfBusClkOn() void { + oracle_sdmmc_bus_clock(1); +} +fn ourBusClkOn() void { + hal.clkrst.setClockEnabled(.sdmmc, true); +} + +fn idfHostDiv4() void { + oracle_sdmmc_set_host_clock_div(4); +} +fn ourHostDiv4() void { + hal.sdmmc.setHostClockDiv(4); +} +fn idfHostDiv8() void { + oracle_sdmmc_set_host_clock_div(8); +} +fn ourHostDiv8() void { + hal.sdmmc.setHostClockDiv(8); +} +fn idfHostDiv10() void { + oracle_sdmmc_set_host_clock_div(10); +} +fn ourHostDiv10() void { + hal.sdmmc.setHostClockDiv(10); +} +fn idfSelectSrc() void { + oracle_sdmmc_select_clk_source_pll160m(); +} +fn ourSelectSrc() void { + hal.sdmmc.selectPll160m(); +} +fn idfPhase() void { + oracle_sdmmc_init_phase_delay(); +} +fn ourPhase() void { + hal.sdmmc.initPhaseDelay(); +} diff --git a/src/oracle/sdmmc_ref.c b/src/oracle/sdmmc_ref.c new file mode 100644 index 0000000..4cc1e59 --- /dev/null +++ b/src/oracle/sdmmc_ref.c @@ -0,0 +1,262 @@ +/* SDMMC's reference half: ESP-IDF's own code, compiled into this image. + * + * Most of what follows is a one-line wrapper over a `sdmmc_ll_*` function, for the same reason + * `gpio_ref.c`'s are: the LL functions are `static inline`, so Zig cannot call them until + * something gives them external linkage, and anything clever here would be a third implementation + * to doubt. + * + * Three of them are not wrappers, and it is worth being explicit about which and why. + * + * 1. `oracle_sdmmc_set_fifo_threshold` writes SDHOST_FIFOTH through `SDMMC.fifoth`. There is no + * `sdmmc_ll` function for that register - ESP-IDF never writes it, on any target - so there + * is nothing to wrap. Writing it through IDF's own bitfield union still makes the bit + * positions IDF's, which is the property the comparison needs. + * + * 2. `oracle_sdmmc_stage_command` transcribes `make_hw_cmd` (sd_trans_sdmmc.c:190-229) and the + * three fields `sd_host_slot_start_command` adds afterwards (sd_host_sdmmc.c:859-881). + * `make_hw_cmd` is `static` in a `.c` file and unreachable from a header, so this is the one + * place the reference is a transcription rather than a call. It is a transcription *into + * IDF's `sdmmc_hw_cmd_t`*, so every bit position still comes from + * `soc/sdmmc_struct.h:354-485` and not from this file; what is being compared is whether the + * Zig side's `Field.of` shifts land in the same places, which is exactly the kind of + * transcription error the oracle exists to catch. + * + * It stages the word with `start_command` cleared. Bit 31 is what launches a command, so a + * staged word is inert: the register can be photographed without the CIU trying to talk to a + * radio that is still in reset. + * + * 3. `oracle_sdmmc_configure_controller` and `oracle_sdmmc_module_reset` are short LL sequences, + * in the order `sd_host_sdmmc.c` performs them. Sequences are the part of a driver that + * register macros cannot express, so a reference for one has to be a sequence too - + * `gpio_ref.c`'s `oracle_gpio_matrix_out` is the same shape. + */ + +/* IDF's clock-and-reset LL functions are shadowed by a macro that references + * `__DECLARE_RCC_ATOMIC_ENV`, an identifier IDF never defines anywhere; its purpose is to make an + * unguarded call fail to compile, because the only legal caller holds a spinlock. There is no + * FreeRTOS here and core 1 is held in reset at power-on, so declaring the name is exactly as safe + * as the spinlock would be. Same as gpio_ref.c and clkrst_ref.c. */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); + +/* `sdmmc_ll_set_command` (sdmmc_ll.h:693-696) calls `memcpy`, and this build's `string.h` is an + * empty stand-in - the register headers need the name to exist, not its contents. Declaring it + * here is enough; the symbol comes from compiler_rt at link time, and at -O2 clang turns a 4-byte + * copy into a single store anyway. */ +#include <stdlib.h> /* the shim's `size_t` */ +void *memcpy(void *dst, const void *src, size_t n); + +#include "hal/sdmmc_ll.h" +#include "soc/sdmmc_struct.h" + +/* ------------------------------------------------------------------ clocks and reset */ + +void oracle_sdmmc_bus_clock(int enable) +{ + sdmmc_ll_enable_bus_clock(0, enable != 0); +} + +void oracle_sdmmc_reset_register(void) +{ + sdmmc_ll_reset_register(0); +} + +void oracle_sdmmc_set_host_clock_div(unsigned div) +{ + sdmmc_ll_set_clock_div(&SDMMC, div); +} + +void oracle_sdmmc_select_clk_source_pll160m(void) +{ + sdmmc_ll_select_clk_source(&SDMMC, SDMMC_CLK_SRC_PLL160M); +} + +void oracle_sdmmc_init_phase_delay(void) +{ + sdmmc_ll_init_phase_delay(&SDMMC); +} + +void oracle_sdmmc_set_card_clock_div(unsigned slot, unsigned div) +{ + sdmmc_ll_set_card_clock_div(&SDMMC, slot, div); +} + +void oracle_sdmmc_enable_card_clock(unsigned slot, int enable) +{ + sdmmc_ll_enable_card_clock(&SDMMC, slot, enable != 0); +} + +void oracle_sdmmc_enable_card_clock_low_power(unsigned slot, int enable) +{ + sdmmc_ll_enable_card_clock_low_power(&SDMMC, slot, enable != 0); +} + +/* ------------------------------------------------------------------ controller resets */ + +void oracle_sdmmc_reset_controller(void) +{ + sdmmc_ll_reset_controller(&SDMMC); +} + +void oracle_sdmmc_reset_dma(void) +{ + sdmmc_ll_reset_dma(&SDMMC); +} + +void oracle_sdmmc_reset_fifo(void) +{ + sdmmc_ll_reset_fifo(&SDMMC); +} + +/* `s_module_reset` plus its completion poll, sd_host_sdmmc.c:917-950. The poll is what makes this + * comparable with the Zig side, which also waits: without it the two could be photographed at + * different points in a self-clearing bit's life. */ +void oracle_sdmmc_module_reset(void) +{ + sdmmc_ll_reset_controller(&SDMMC); + sdmmc_ll_reset_dma(&SDMMC); + sdmmc_ll_reset_fifo(&SDMMC); + while (!(sdmmc_ll_is_controller_reset_done(&SDMMC) && + sdmmc_ll_is_dma_reset_done(&SDMMC) && + sdmmc_ll_is_fifo_reset_done(&SDMMC))) { + /* bounded by the caller: the harness runs this with the bus clock on, where the three bits + * clear in a handful of cycles. */ + } +} + +/* ------------------------------------------------------------------ transfer geometry */ + +void oracle_sdmmc_set_card_width(unsigned slot, unsigned width) +{ + sdmmc_ll_set_card_width(&SDMMC, slot, + width == 4 ? SD_BUS_WIDTH_4_BIT : SD_BUS_WIDTH_1_BIT); +} + +void oracle_sdmmc_set_block_size(unsigned size) +{ + sdmmc_ll_set_block_size(&SDMMC, size); +} + +void oracle_sdmmc_set_data_transfer_len(unsigned len) +{ + sdmmc_ll_set_data_transfer_len(&SDMMC, len); +} + +void oracle_sdmmc_set_timeouts(unsigned data_cycles, unsigned response_cycles) +{ + sdmmc_ll_set_data_timeout(&SDMMC, data_cycles); + sdmmc_ll_set_response_timeout(&SDMMC, response_cycles); +} + +/* No `sdmmc_ll` function exists for this register; see note 1 at the head of the file. */ +void oracle_sdmmc_set_fifo_threshold(unsigned rx_wmark, unsigned tx_wmark, unsigned msize) +{ + SDMMC.fifoth.rx_wmark = rx_wmark; + SDMMC.fifoth.tx_wmark = tx_wmark; + SDMMC.fifoth.dma_multiple_transaction_size = msize; +} + +/* ------------------------------------------------------------------ interrupts and DMA */ + +/* sd_host_sdmmc.c:120-124, in order: clear everything, mask everything, global off, unmask the + * default set, global on - and then the one thing this project does that ESP-IDF does not: mask + * and clear card detect. + * + * That last pair is a deliberate deviation, so it is expressed here through IDF's own LL rather + * than left to differ. There is no card-detect pin on this board; `configurePins` ties the signal + * to a matrix constant, the transition latches RINTSTS.cd, and nothing in the command path clears + * bit 0 - so an unmasked cd holds the controller's line into the CLIC high forever. Comparing an + * IDF sequence that leaves it unmasked against a Zig one that does not would report a difference + * that is the point rather than a bug; comparing the same intent on both sides still catches a + * wrong bit, a wrong register or a wrong order. */ +void oracle_sdmmc_configure_interrupts(void) +{ + sdmmc_ll_clear_interrupt(&SDMMC, 0xffffffff); + sdmmc_ll_enable_interrupt(&SDMMC, 0xffffffff, false); + sdmmc_ll_enable_global_interrupt(&SDMMC, false); + sdmmc_ll_enable_interrupt(&SDMMC, SDMMC_LL_EVENT_DEFAULT, true); + sdmmc_ll_enable_interrupt(&SDMMC, SDMMC_LL_EVENT_CD, false); + sdmmc_ll_clear_interrupt(&SDMMC, SDMMC_LL_EVENT_CD); + sdmmc_ll_enable_global_interrupt(&SDMMC, true); +} + +void oracle_sdmmc_init_dma(void) +{ + sdmmc_ll_init_dma(&SDMMC); +} + +void oracle_sdmmc_enable_dma(int enable) +{ + sdmmc_ll_enable_dma(&SDMMC, enable != 0); +} + +void oracle_sdmmc_set_desc_addr(unsigned addr) +{ + sdmmc_ll_set_desc_addr(&SDMMC, addr); +} + +void oracle_sdmmc_enable_sdio_interrupt(unsigned slot, int enable) +{ + sdmmc_ll_enable_interrupt(&SDMMC, slot == 0 ? SDMMC_LL_EVENT_IO_SLOT0 : SDMMC_LL_EVENT_IO_SLOT1, + enable != 0); +} + +/* ------------------------------------------------------------------ the command word */ + +/* make_hw_cmd (sd_trans_sdmmc.c:190-229) + sd_host_slot_start_command's three additions + * (sd_host_sdmmc.c:859-881), staged with start_command cleared. See note 2 at the head of the + * file for why this one is a transcription. + * + * `data` is 0 for none, 1 for read, 2 for write - the same three-way choice `cmd->data` and + * `SCF_CMD_READ` encode between them. */ +void oracle_sdmmc_stage_command(unsigned index, int response_long, int response_expect, + int check_crc, int data, int send_init, int wait_prvdata, + int update_clk, unsigned slot) +{ + sdmmc_hw_cmd_t res = { 0 }; + + res.cmd_index = index; + if (send_init) { + res.send_init = 1; + } + if (wait_prvdata) { + res.wait_complete = 1; + } + if (response_expect) { + res.response_expect = 1; + if (response_long) { + res.response_long = 1; + } + } + if (check_crc) { + res.check_response_crc = 1; + } + if (data) { + res.data_expected = 1; + if (data == 2) { + res.rw = 1; + } + } + if (update_clk) { + res.update_clk_reg = 1; + } + + /* sd_host_slot_start_command: "Outputs should be synchronized to cclk_out". */ + res.use_hold_reg = 1; + res.card_num = slot; + /* Deliberately *not* res.start_command = 1: staging, not sending. */ + res.start_command = 0; + + sdmmc_ll_set_command(&SDMMC, res); +} + +/* ------------------------------------------------------------------ observation */ + +unsigned oracle_sdmmc_version_id(void) +{ + return sdmmc_ll_get_version_id(&SDMMC); +} + +unsigned oracle_sdmmc_hw_config(void) +{ + return sdmmc_ll_get_hw_config_info(&SDMMC); +} diff --git a/src/oracle/timg_cases.zig b/src/oracle/timg_cases.zig new file mode 100644 index 0000000..34a31de --- /dev/null +++ b/src/oracle/timg_cases.zig @@ -0,0 +1,435 @@ +//! TIMG's side of the differential test: the same timer and watchdog operations expressed as +//! ESP-IDF's LL calls and as this project's HAL calls. +//! +//! **TIMG1 throughout, never TIMG0.** TIMG0 hosts MWDT0, the watchdog the rest of the system relies +//! on staying quiet; this image's bootloader has already disabled it. A mistake in a case that ran +//! against group 0 would not fail a comparison, it would reboot the board mid-run with nothing on +//! the console to explain it. +//! +//! The block is restored by `configure` rather than by the harness pulsing a `reset_bit`, and the +//! reason is the whole safety story of this peripheral: resetting a timer group re-arms +//! `WDT_FLASHBOOT_MOD_EN`, which runs the watchdog independently of `WDT_EN`, so a bare reset-bit +//! pulse arms a watchdog nobody is feeding. `clkrst.resetPeripheral(.timg1)` pulses the bit *and* +//! clears that flag - exactly as `_timg_ll_reset_register` does (timg_ll.h:60-71) - and the harness's +//! `reset_bit` path does only the pulse. So the restore goes through the HAL, and the reset sequence +//! itself becomes one of the cases below instead. +//! +//! The restore deliberately leaves the watchdog **write-protected**. That makes the unlock half of +//! every watchdog case load-bearing: an implementation that forgot to lift protection would have its +//! stage and prescaler writes silently dropped and would differ from IDF's in the snapshot, rather +//! than passing because both sides happened to be unlocked already. +//! +//! What is *not* here, and why: the timers' function-clock source and per-timer gate live in +//! HP_SYS_CLKRST (PERI_CLK_CTRL20/21), and the group's bus-clock gate in SOC_CLK_CTRL2, none of +//! which is inside this block. The harness compares one contiguous window of at most 512 words and +//! HP_SYS_CLKRST is ~0x1e000 bytes away from TIMG1, so a gate case here would compare two identical +//! TIMG snapshots and pass no matter what it wrote. Those pairings need a HP_SYS_CLKRST suite of +//! their own; the reset case below is the one part of that story this window can see, and it does +//! see it, because a group reset and the flashboot fixup both land in these 64 words. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +const timg = hal.timg; + +// ------------------------------------------------------------------- ESP-IDF's side, from timg_ref.c + +extern fn oracle_timg_set_divider(group: c_int, timer: c_uint, divider: c_uint) void; +extern fn oracle_timg_set_direction_up(group: c_int, timer: c_uint, up: c_int) void; +extern fn oracle_timg_set_auto_reload(group: c_int, timer: c_uint, en: c_int) void; +extern fn oracle_timg_enable_counter(group: c_int, timer: c_uint, en: c_int) void; +extern fn oracle_timg_enable_alarm(group: c_int, timer: c_uint, en: c_int) void; +extern fn oracle_timg_set_alarm_value(group: c_int, timer: c_uint, value: c_ulonglong) void; +extern fn oracle_timg_set_reload_value(group: c_int, timer: c_uint, value: c_ulonglong) void; +extern fn oracle_timg_trigger_soft_reload(group: c_int, timer: c_uint) void; +extern fn oracle_timg_read_counter(group: c_int, timer: c_uint) c_ulonglong; +extern fn oracle_timg_reset_register(group: c_int) void; + +extern fn oracle_mwdt_set_stage(group: c_int, stage: c_uint, timeout: c_uint, action: c_uint) void; +extern fn oracle_mwdt_disable_stage(group: c_int, stage: c_uint) void; +extern fn oracle_mwdt_set_prescaler(group: c_int, prescaler: c_uint) void; +extern fn oracle_mwdt_set_cpu_reset_length(group: c_int, length: c_uint) void; +extern fn oracle_mwdt_set_sys_reset_length(group: c_int, length: c_uint) void; +extern fn oracle_mwdt_set_flashboot_en(group: c_int, en: c_int) void; +extern fn oracle_mwdt_set_enabled(group: c_int, en: c_int) void; +extern fn oracle_mwdt_feed(group: c_int) void; +extern fn oracle_mwdt_write_protect_disable(group: c_int) void; +extern fn oracle_mwdt_write_protect_enable(group: c_int) void; + +/// The group under test, as a number for the C side. Deliberately a constant rather than a variable: +/// unlike GPIO's pin, this is not a parameter to sweep, it is a safety property. +const group_id: c_int = 1; +const group: timg.Group = .timg1; + +/// The timer under test. A module-level `var` because Zig has no closures and the harness stores +/// plain `fn` pointers; the suite runs the whole list once per timer in `timers`. +pub var timer: timg.Timer = .t0; + +/// Both general-purpose timers of the group (TIMG_LL_GPTIMERS_PER_INST is 2 on the P4). Worth +/// sweeping because the timer index is a *stride* in this HAL rather than a separate set of macros, +/// and a wrong stride writes into the neighbouring timer's registers. +pub const timers = [_]timg.Timer{ .t0, .t1 }; + +inline fn timerId() c_uint { + return @intFromEnum(timer); +} + +// ------------------------------------------------------------------------------------ restore + +fn restore() void { + // Pulses HP_RST_EN1's TIMERGRP1 bit and then clears WDT_FLASHBOOT_MOD_EN, which the pulse + // re-armed. Both halves matter; see the file comment. + // ESP-IDF's reset, not ours: this suite's `reset_register_clears_flashboot` case exists to + // compare the two, and restoring with ours would let a no-op reset pass it. + oracle_timg_reset_register(group_id); + // IDF's reset re-arms flash-boot protection and does not clear it, so clear it here through the + // register directly - the board reboots a few seconds later otherwise. + mmio.Reg.atAddress(@intCast(regs.TIMG_WDTCONFIG0_REG(1))) + .modify(.{mmio.Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V).is(0)}); + // Leave write protection on, so every watchdog case has to lift it itself. + timg.unlock(group).release(); +} + +// ------------------------------------------------------------------------------------- suite + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "timg1", + // TIMG_T0CONFIG_REG is at +0x00 of the group's block (timer_group_reg.h:19), and the group + // stride is 0x1000 (:14). + .base = @intCast(regs.TIMG_T0CONFIG_REG(1)), + // 0x100 bytes: the last register in the block is TIMG_REGCLK_REG at +0xfc. The window has to + // reach it - TIMG_WDTWPROTECT_REG is at +0x64 and the four stage-timeout registers at + // +0x50..+0x5c, so a window that stopped at the timers (+0x48) would be blind to every + // watchdog case in this file. + .words = 64, + .volatile_words = &.{ + (0x04 - 0x00) / 4, // TIMG_T0LO - the captured counter, which moves between snapshots + (0x08 - 0x00) / 4, // TIMG_T0HI + (0x28 - 0x00) / 4, // TIMG_T1LO + (0x2c - 0x00) / 4, // TIMG_T1HI + (0x68 - 0x00) / 4, // TIMG_RTCCALICFG - RTC calibration runs cyclically by default + (0x6c - 0x00) / 4, // TIMG_RTCCALICFG1 - and latches a new count each cycle + (0x74 - 0x00) / 4, // TIMG_INT_RAW_TIMERS - alarm/watchdog raw status, set by hardware + (0x78 - 0x00) / 4, // TIMG_INT_ST_TIMERS + (0x80 - 0x00) / 4, // TIMG_RTCCALICFG2 + }, + // TIMG1's bus clock: SOC_CLK_CTRL2 bit 22 (hp_sys_clkrst_reg.h:763, and timg_ll.h:35-42 + // for the register it belongs to - not PERI_CLK_CTRL21, which is where this project's + // clkrst table had it until this suite was written). A snapshot of a gated block returns + // the last latched value rather than zeros, so the harness checks this first. + .clock = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL2_REG), + .bit = @intCast(regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_S), + }, + .restore = .{ .configure = restore }, + }, + .cases = &.{ + // ---- prescaler. 2 is the hardware minimum and 65536 is the maximum, encoded as 0 + // (timer_ll.h:191-199) - the one arithmetic edge in this peripheral. + .{ .name = "divider", .arg = 2, .idf = idfDivider2, .ours = ourDivider2 }, + .{ .name = "divider", .arg = 1234, .idf = idfDivider1234, .ours = ourDivider1234 }, + .{ .name = "divider", .arg = 65535, .idf = idfDivider65535, .ours = ourDivider65535 }, + .{ .name = "divider_wraps_to_zero", .arg = 65536, .idf = idfDivider65536, .ours = ourDivider65536 }, + // ---- direction, auto-reload, counter and alarm enables + .{ .name = "direction_up", .arg = 1, .idf = idfDirUp, .ours = ourDirUp }, + .{ .name = "direction_down", .arg = 0, .idf = idfDirDown, .ours = ourDirDown }, + .{ .name = "auto_reload_on", .arg = 1, .idf = idfReloadOn, .ours = ourReloadOn }, + .{ .name = "auto_reload_off", .arg = 0, .idf = idfReloadOff, .ours = ourReloadOff }, + .{ .name = "counter_enable", .arg = 1, .idf = idfCounterOn, .ours = ourCounterOn }, + .{ .name = "counter_disable", .arg = 0, .idf = idfCounterOff, .ours = ourCounterOff }, + .{ .name = "alarm_enable", .arg = 1, .idf = idfAlarmOn, .ours = ourAlarmOn }, + .{ .name = "alarm_disable", .arg = 0, .idf = idfAlarmOff, .ours = ourAlarmOff }, + // ---- the 54-bit pairs. 0x2a_5555_aaaa exercises all 22 bits of the high word: a value + // that fit in 32 bits would pass even if the high half were dropped entirely. + .{ .name = "alarm_value_54bit", .arg = 0x5555_aaaa, .idf = idfAlarmValue, .ours = ourAlarmValue }, + .{ .name = "alarm_value_zero", .arg = 0, .idf = idfAlarmValueZero, .ours = ourAlarmValueZero }, + .{ .name = "load_value_54bit", .arg = 0x1234_5678, .idf = idfLoadValue, .ours = ourLoadValue }, + // Write-to-trigger: nothing in the compared window changes, and the counter registers are + // volatile. The case is here because it would catch the trigger landing on the wrong + // address - TIMG_T0LOAD_REG is one word past TIMG_T0LOADHI_REG - which is a live risk when + // the timer index is a stride rather than a distinct macro. + .{ .name = "soft_reload_trigger", .idf = idfSoftReload, .ours = ourSoftReload }, + // The latch-then-read sequence. Register-identical by construction, so what it really + // proves is that our poll terminates: this peripheral acknowledges a capture by *clearing* + // TxUPDATE, and waiting for it to be set instead hangs the run. + .{ .name = "read_counter_latch", .idf = idfReadCounter, .ours = ourReadCounter }, + // ---- watchdog. Every one of these has to lift write protection and put it back; the + // restored state has it on, so a dropped unlock shows up as a difference. + .{ .name = "wdt_write_protect_dance", .idf = idfWdtDance, .ours = ourWdtDance }, + .{ .name = "wdt_stage0_interrupt", .arg = 2_000_000, .idf = idfWdtStage0, .ours = ourWdtStage0 }, + .{ .name = "wdt_stage1_reset_cpu", .arg = 5_000, .idf = idfWdtStage1, .ours = ourWdtStage1 }, + .{ .name = "wdt_stage2_reset_system", .arg = 123_456, .idf = idfWdtStage2, .ours = ourWdtStage2 }, + .{ .name = "wdt_stage3_off", .idf = idfWdtStage3Off, .ours = ourWdtStage3Off }, + .{ .name = "wdt_prescaler", .arg = 20_000, .idf = idfWdtPrescaler, .ours = ourWdtPrescaler }, + .{ .name = "wdt_cpu_reset_length", .arg = 7, .idf = idfWdtCpuLen, .ours = ourWdtCpuLen }, + .{ .name = "wdt_sys_reset_length", .arg = 4, .idf = idfWdtSysLen, .ours = ourWdtSysLen }, + .{ .name = "wdt_flashboot_off", .arg = 0, .idf = idfWdtFlashbootOff, .ours = ourWdtFlashbootOff }, + .{ .name = "wdt_feed", .idf = idfWdtFeed, .ours = ourWdtFeed }, + // Safe on TIMG1 only because the restored state has all four stages off and flashboot mode + // cleared, so an enabled watchdog here has no action to take before the next restore. + .{ .name = "wdt_enable", .arg = 1, .idf = idfWdtEnable, .ours = ourWdtEnable }, + .{ .name = "wdt_disable", .arg = 0, .idf = idfWdtDisable, .ours = ourWdtDisable }, + // ---- the reset sequence itself, which is the only part of the clock/reset table this + // window can see: the group reset plus the flashboot fixup that has to follow it. + .{ .name = "reset_register_clears_flashboot", .idf = idfResetRegister, .ours = ourResetRegister }, + }, + .setup = setup, +}; + +/// The group's bus clock. Already 1 out of reset (hp_sys_clkrst_reg.h:763, default 1) and this image +/// never runs `esp_perip_clk_init`, so this is belt-and-braces - but a snapshot of a gated block is +/// stale rather than zero, and the harness would rather fail the gate check than compare noise. +fn setup() void { + hal.clkrst.setClockEnabled(.timg1, true); +} + +// -------------------------------------------------------------------------- the case pairs +// Same operation, same arguments, twice. IDF's LL on one side, this HAL on the other; a read-back +// through our own accessor would prove nothing, which is the whole point of the arrangement. + +fn idfDivider2() void { + oracle_timg_set_divider(group_id, timerId(), 2); +} +fn ourDivider2() void { + timg.setDivider(group, timer, 2); +} +fn idfDivider1234() void { + oracle_timg_set_divider(group_id, timerId(), 1234); +} +fn ourDivider1234() void { + timg.setDivider(group, timer, 1234); +} +fn idfDivider65535() void { + oracle_timg_set_divider(group_id, timerId(), 65535); +} +fn ourDivider65535() void { + timg.setDivider(group, timer, 65535); +} +fn idfDivider65536() void { + oracle_timg_set_divider(group_id, timerId(), 65536); +} +fn ourDivider65536() void { + timg.setDivider(group, timer, 65536); +} + +fn idfDirUp() void { + oracle_timg_set_direction_up(group_id, timerId(), 1); +} +fn ourDirUp() void { + timg.setDirection(group, timer, .up); +} +fn idfDirDown() void { + oracle_timg_set_direction_up(group_id, timerId(), 0); +} +fn ourDirDown() void { + timg.setDirection(group, timer, .down); +} + +fn idfReloadOn() void { + oracle_timg_set_auto_reload(group_id, timerId(), 1); +} +fn ourReloadOn() void { + timg.setAutoReload(group, timer, true); +} +fn idfReloadOff() void { + oracle_timg_set_auto_reload(group_id, timerId(), 0); +} +fn ourReloadOff() void { + timg.setAutoReload(group, timer, false); +} + +fn idfCounterOn() void { + oracle_timg_enable_counter(group_id, timerId(), 1); +} +fn ourCounterOn() void { + timg.setCounterEnabled(group, timer, true); +} +fn idfCounterOff() void { + oracle_timg_enable_counter(group_id, timerId(), 0); +} +fn ourCounterOff() void { + timg.setCounterEnabled(group, timer, false); +} + +fn idfAlarmOn() void { + oracle_timg_enable_alarm(group_id, timerId(), 1); +} +fn ourAlarmOn() void { + timg.setAlarmEnabled(group, timer, true); +} +fn idfAlarmOff() void { + oracle_timg_enable_alarm(group_id, timerId(), 0); +} +fn ourAlarmOff() void { + timg.setAlarmEnabled(group, timer, false); +} + +/// 54 bits: 22 in the high word, 32 in the low one. +const alarm_value: u64 = 0x2a_5555_aaaa; +const load_value: u64 = 0x15_1234_5678; + +fn idfAlarmValue() void { + oracle_timg_set_alarm_value(group_id, timerId(), alarm_value); +} +fn ourAlarmValue() void { + timg.setAlarmValue(group, timer, alarm_value); +} +fn idfAlarmValueZero() void { + oracle_timg_set_alarm_value(group_id, timerId(), 0); +} +fn ourAlarmValueZero() void { + timg.setAlarmValue(group, timer, 0); +} +fn idfLoadValue() void { + oracle_timg_set_reload_value(group_id, timerId(), load_value); +} +fn ourLoadValue() void { + timg.setLoadValue(group, timer, load_value); +} +fn idfSoftReload() void { + oracle_timg_set_reload_value(group_id, timerId(), load_value); + oracle_timg_trigger_soft_reload(group_id, timerId()); +} +fn ourSoftReload() void { + timg.setLoadValue(group, timer, load_value); + timg.load(group, timer); +} + +fn idfReadCounter() void { + _ = oracle_timg_read_counter(group_id, timerId()); +} +fn ourReadCounter() void { + // Discarding the value is the point: the comparison is over registers, and what this exercises + // is the handshake. A null return means our poll gave up after 10,000 reads, which IDF's + // version cannot report because it spins forever. + _ = timg.read(group, timer); +} + +// ------------------------------------------------------------------------------ watchdog pairs + +fn idfWdtDance() void { + oracle_mwdt_write_protect_disable(group_id); + oracle_mwdt_write_protect_enable(group_id); +} +fn ourWdtDance() void { + const wdt = timg.unlock(group); + wdt.release(); +} + +fn idfWdtStage0() void { + oracle_mwdt_set_stage(group_id, 0, 2_000_000, @intFromEnum(timg.Action.interrupt)); +} +fn ourWdtStage0() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setStage(.stage0, 2_000_000, .interrupt); +} + +fn idfWdtStage1() void { + oracle_mwdt_set_stage(group_id, 1, 5_000, @intFromEnum(timg.Action.reset_cpu)); +} +fn ourWdtStage1() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setStage(.stage1, 5_000, .reset_cpu); +} + +fn idfWdtStage2() void { + oracle_mwdt_set_stage(group_id, 2, 123_456, @intFromEnum(timg.Action.reset_system)); +} +fn ourWdtStage2() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setStage(.stage2, 123_456, .reset_system); +} + +fn idfWdtStage3Off() void { + // Configure it to something first, so "off" has something to undo and the case cannot pass by + // both sides doing nothing. + oracle_mwdt_set_stage(group_id, 3, 999, @intFromEnum(timg.Action.interrupt)); + oracle_mwdt_disable_stage(group_id, 3); +} +fn ourWdtStage3Off() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setStage(.stage3, 999, .interrupt); + wdt.disableStage(.stage3); +} + +fn idfWdtPrescaler() void { + oracle_mwdt_set_prescaler(group_id, 20_000); +} +fn ourWdtPrescaler() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setPrescaler(20_000); +} + +fn idfWdtCpuLen() void { + oracle_mwdt_set_cpu_reset_length(group_id, @intFromEnum(timg.ResetLength.us_3_2)); +} +fn ourWdtCpuLen() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setCpuResetLength(.us_3_2); +} + +fn idfWdtSysLen() void { + oracle_mwdt_set_sys_reset_length(group_id, @intFromEnum(timg.ResetLength.ns_500)); +} +fn ourWdtSysLen() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setSysResetLength(.ns_500); +} + +fn idfWdtFlashbootOff() void { + oracle_mwdt_set_flashboot_en(group_id, 0); +} +fn ourWdtFlashbootOff() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setFlashbootEnabled(false); +} + +fn idfWdtFeed() void { + oracle_mwdt_feed(group_id); +} +fn ourWdtFeed() void { + timg.feed(group); +} + +fn idfWdtEnable() void { + oracle_mwdt_set_enabled(group_id, 1); +} +fn ourWdtEnable() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setEnabled(true); +} + +fn idfWdtDisable() void { + oracle_mwdt_set_enabled(group_id, 0); +} +fn ourWdtDisable() void { + const wdt = timg.unlock(group); + defer wdt.release(); + wdt.setEnabled(false); +} + +fn idfResetRegister() void { + oracle_timg_reset_register(group_id); +} +fn ourResetRegister() void { + // ESP-IDF's reset, not ours: this suite's `reset_register_clears_flashboot` case exists to + // compare the two, and restoring with ours would let a no-op reset pass it. + oracle_timg_reset_register(group_id); + // IDF's reset re-arms flash-boot protection and does not clear it, so clear it here through the + // register directly - the board reboots a few seconds later otherwise. + mmio.Reg.atAddress(@intCast(regs.TIMG_WDTCONFIG0_REG(1))) + .modify(.{mmio.Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V).is(0)}); +} diff --git a/src/oracle/timg_ref.c b/src/oracle/timg_ref.c new file mode 100644 index 0000000..38bccc8 --- /dev/null +++ b/src/oracle/timg_ref.c @@ -0,0 +1,197 @@ +/* The reference implementation for the timer groups and their watchdogs, which is ESP-IDF's own. + * + * Thin external-linkage wrappers over `timer_ll.h`, `mwdt_ll.h` and `timg_ll.h`, so Zig can call + * IDF's `static inline` functions and the differential harness can run both implementations in one + * image on one boot. There is no logic here: anything clever would be a third implementation to + * doubt. + * + * Two things about the watchdog wrappers are deliberate. The write-protect dance is *inside* each + * wrapper (`mwdt_ll_write_protect_disable` ... `mwdt_ll_write_protect_enable`) because that is what + * IDF's callers do - `mwdt_ll_config_stage` itself will silently do nothing if protection is on - + * and because our side does the same thing through `timg.unlock`/`release`. The two sides have to be + * the same operation, key register included, or comparing WDTWPROTECT afterwards means nothing. + * And nothing here ever calls `mwdt_ll_enable` on group 0: TIMG0 hosts the watchdog the rest of the + * system depends on not firing. + */ + +/* IDF's clock and reset LL functions are shadowed by a wrapper macro referencing + * `__DECLARE_RCC_ATOMIC_ENV` / `__DECLARE_RCC_RC_ATOMIC_ENV`, identifiers IDF never defines + * anywhere: their purpose is to make an unguarded call fail to compile, because the only legal + * caller holds a FreeRTOS spinlock. There is no FreeRTOS here and core 1 is held in reset at + * power-on, so declaring the names is exactly as safe as the spinlock would be - and it is what + * IDF's own bootloader does (bootloader_support/src/bootloader_console.c:53). */ +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); +static int __DECLARE_RCC_RC_ATOMIC_ENV __attribute__((unused)); + +#include "hal/timer_ll.h" +#include "hal/mwdt_ll.h" +#include "hal/timg_ll.h" +#include "soc/timer_group_struct.h" + +static timg_dev_t *grp(int group) +{ + return TIMER_LL_GET_HW(group); +} + +/* ------------------------------------------------------------------ general purpose timer */ + +void oracle_timg_set_divider(int group, unsigned timer, unsigned divider) +{ + timer_ll_set_clock_prescale(grp(group), timer, divider); +} + +void oracle_timg_set_direction_up(int group, unsigned timer, int up) +{ + timer_ll_set_count_direction(grp(group), timer, up ? GPTIMER_COUNT_UP : GPTIMER_COUNT_DOWN); +} + +void oracle_timg_set_auto_reload(int group, unsigned timer, int en) +{ + timer_ll_enable_auto_reload(grp(group), timer, en != 0); +} + +void oracle_timg_enable_counter(int group, unsigned timer, int en) +{ + timer_ll_enable_counter(grp(group), timer, en != 0); +} + +void oracle_timg_enable_alarm(int group, unsigned timer, int en) +{ + timer_ll_enable_alarm(grp(group), timer, en != 0); +} + +void oracle_timg_set_alarm_value(int group, unsigned timer, unsigned long long value) +{ + timer_ll_set_alarm_value(grp(group), timer, value); +} + +void oracle_timg_set_reload_value(int group, unsigned timer, unsigned long long value) +{ + timer_ll_set_reload_value(grp(group), timer, value); +} + +unsigned long long oracle_timg_get_reload_value(int group, unsigned timer) +{ + return timer_ll_get_reload_value(grp(group), timer); +} + +void oracle_timg_trigger_soft_reload(int group, unsigned timer) +{ + timer_ll_trigger_soft_reload(grp(group), timer); +} + +/* The latch-then-read sequence: `timer_ll_trigger_soft_capture` writes TxUPDATE and spins until the + * hardware clears it, and only then is the TxHI/TxLO pair meaningful. Exposed as one call because + * that is how our `timg.read` expresses it, and splitting it would compare halves of a sequence. */ +unsigned long long oracle_timg_read_counter(int group, unsigned timer) +{ + timer_ll_trigger_soft_capture(grp(group), timer); + return timer_ll_get_counter_value(grp(group), timer); +} + +void oracle_timg_set_clock_source_xtal(int group, unsigned timer) +{ + timer_ll_set_clock_source(group, timer, GPTIMER_CLK_SRC_XTAL); +} + +void oracle_timg_set_clock_source_pll80m(int group, unsigned timer) +{ + timer_ll_set_clock_source(group, timer, GPTIMER_CLK_SRC_PLL_F80M); +} + +void oracle_timg_enable_timer_clock(int group, unsigned timer, int en) +{ + timer_ll_enable_clock(group, timer, en != 0); +} + +void oracle_timg_enable_bus_clock(int group, int en) +{ + timg_ll_enable_bus_clock(group, en != 0); +} + +/* Pulses the group's reset bit and then clears WDT_FLASHBOOT_MOD_EN, which the reset re-arms + * (timg_ll.h:51-71). The clearing is the interesting half: leave it out and the board reboots a + * moment later with nothing on the console to explain it. */ +void oracle_timg_reset_register(int group) +{ + timg_ll_reset_register(group); +} + +/* --------------------------------------------------------------------------------- watchdog */ + +void oracle_mwdt_set_stage(int group, unsigned stage, unsigned timeout, unsigned action) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_config_stage(grp(group), (wdt_stage_t)stage, timeout, (wdt_stage_action_t)action); + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_disable_stage(int group, unsigned stage) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_disable_stage(grp(group), stage); + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_set_prescaler(int group, unsigned prescaler) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_set_prescaler(grp(group), prescaler); + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_set_cpu_reset_length(int group, unsigned length) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_set_cpu_reset_length(grp(group), (wdt_reset_sig_length_t)length); + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_set_sys_reset_length(int group, unsigned length) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_set_sys_reset_length(grp(group), (wdt_reset_sig_length_t)length); + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_set_flashboot_en(int group, int en) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_set_flashboot_en(grp(group), en != 0); + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_set_enabled(int group, int en) +{ + mwdt_ll_write_protect_disable(grp(group)); + if (en) { + mwdt_ll_enable(grp(group)); + } else { + mwdt_ll_disable(grp(group)); + } + mwdt_ll_write_protect_enable(grp(group)); +} + +void oracle_mwdt_feed(int group) +{ + mwdt_ll_write_protect_disable(grp(group)); + mwdt_ll_feed(grp(group)); + mwdt_ll_write_protect_enable(grp(group)); +} + +/* The two halves of the protection dance on their own, so a case can check that our key value and + * IDF's are the same word rather than only that a guarded sequence ends up locked. */ +void oracle_mwdt_write_protect_disable(int group) +{ + mwdt_ll_write_protect_disable(grp(group)); +} + +void oracle_mwdt_write_protect_enable(int group) +{ + mwdt_ll_write_protect_enable(grp(group)); +} + +int oracle_mwdt_is_enabled(int group) +{ + return mwdt_ll_check_if_enabled(grp(group)) ? 1 : 0; +} diff --git a/src/oracle/uart_cases.zig b/src/oracle/uart_cases.zig new file mode 100644 index 0000000..444d235 --- /dev/null +++ b/src/oracle/uart_cases.zig @@ -0,0 +1,326 @@ +//! UART's side of the differential test. +//! +//! **The peripheral under test is UART1, and that is a safety constraint rather than a preference.** +//! UART0 carries this board's console. The harness restores a UART by pulsing its reset bit, and +//! resetting UART0 clears UART_CLKDIV: the console's output turns to garbage mid-character and the +//! board takes a watchdog reset with nothing readable left to explain it. That was measured on this +//! board. UART1 is otherwise idle here, has no pins routed at power-on, and resets cleanly. +//! +//! **Word 0 is on the no-read list.** `UART_FIFO_REG` is at offset 0x000 - the first word any "read +//! the whole block" loop touches - its only field is annotated `RO` in uart_reg.h:18, and that +//! annotation is wrong in the way that matters: the read is the FIFO pop. A generic snapshot of a +//! UART eats received bytes. +//! +//! **What this suite cannot see, stated plainly.** The descriptor is one contiguous window and the +//! UART's is 0xa0 bytes at its own base, so three things this HAL does land outside it: +//! +//! * the integer pre-divider `REG_UART1_SCLK_DIV_NUM` and the source select +//! `REG_UART1_CLK_SRC_SEL`, which are in HP_SYS_CLKRST at a different base; +//! * the GPIO matrix registers the routing cases write, which are in the GPIO block; +//! * the FIFO contents themselves, which have no addressable state to compare. +//! +//! The pre-divider is not unobserved, though, only observed indirectly: `clk_div` is +//! `(sclk_freq << 4) / (baud * sclk_div)`, so the in-window CLKDIV_SYNC word is a function of the +//! pre-divider, and the two sides disagreeing on `sclk_div` shows up as a different CLKDIV unless +//! the two errors cancel exactly. The `baud_300` case exists specifically because it is the one +//! rate here whose pre-divider is not 1. The routing cases are honestly weak in this window - what +//! they compare is that both sides leave the *UART* untouched, and their real evidence is that +//! `gpio_cases`' `matrix_out` case passes against the same GPIO LL functions this file calls. +//! +//! Both sides reach the hardware by different paths throughout: the `idf` half calls ESP-IDF's +//! `uart_ll.h` compiled by clang, the `ours` half calls src/hal/uart.zig. Nothing here reads a +//! value back through the accessor that wrote it, because that proves only that the accessor is +//! self-consistent. + +const std = @import("std"); +const hal = @import("hal"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const types = @import("differ_types.zig"); + +extern fn oracle_uart_set_sclk(num: c_uint, sel: c_uint) void; +/// The source select, named on the C side. `UART_SCLK_XTAL` is a `soc_module_clk_t` enumerator whose +/// numeric value is an accident of a chip-wide enum, so it must not cross this boundary as an +/// integer - passing 0 selects nothing that exists and hangs the next commit. +extern fn oracle_uart_set_sclk_xtal(num: c_uint) void; +extern fn oracle_uart_sclk_enable(num: c_uint) void; +extern fn oracle_uart_enable_bus_clock(num: c_uint, enable: c_int) void; +extern fn oracle_uart_set_baudrate(num: c_uint, baud: c_uint, sclk_freq: c_uint) c_int; +extern fn oracle_uart_set_data_bit_num(num: c_uint, bits: c_uint) void; +extern fn oracle_uart_set_stop_bits(num: c_uint, stop: c_uint) void; +extern fn oracle_uart_set_parity(num: c_uint, parity: c_uint) void; +extern fn oracle_uart_txfifo_rst(num: c_uint) void; +extern fn oracle_uart_rxfifo_rst(num: c_uint) void; +extern fn oracle_uart_set_loop_back(num: c_uint, enable: c_int) void; +extern fn oracle_uart_update(num: c_uint) void; +extern fn oracle_uart_route_tx(num: c_uint, pin: c_uint) void; +extern fn oracle_uart_route_rx(num: c_uint, pin: c_uint) void; + +/// The instance under test. A module-level `var` because Zig has no closures and the harness stores +/// plain `fn` pointers. It is a `var` rather than a constant so a future run can move to UART2-4, +/// but it must never become 0: see this file's header. +pub var port: u8 = 1; + +/// The pad the routing cases use. GPIO33 is a free pin on this board's JP1 header - the same one +/// `gpio_cases` uses for its high-bank tests, and for the same reason. +pub var route_pin: u8 = 33; + +/// The clock source frequency the baud cases assume, matching what `setup` selects. XTAL is 40 MHz +/// on the P4 and is the only source whose frequency is exact, which is what makes an expected +/// divider computable by hand. +const sclk_freq: u32 = 40_000_000; + +fn ours() hal.uart.Uart { + return hal.uart.Uart.init(port); +} + +/// Bring UART1 far enough up that its registers answer and its baud generator runs: APB bus clock, +/// core clock, and a source select. Done through IDF's LL rather than ours, so that a bug in our +/// clock code cannot make the whole suite silently compare two dead blocks - and the harness +/// re-checks the bus clock gate before every case regardless. +fn setup() void { + restore(); +} + +/// Known state: out of reset, bus clock on, core clock on, source selected. Every case starts here. +/// +/// The reset is what makes this a sound restore for a block whose CONF0_SYNC carries two +/// write-to-act FIFO resets and whose offset 0 transmits when written - there is nothing here that +/// could be restored by writing a saved snapshot back. The re-enable is what makes it *usable* +/// afterwards. +fn restore() void { + const guard = hal.clkrst.maskInterrupts(); + const rst = mmio.Reg.at(regs.HP_SYS_CLKRST_HP_RST_EN1_REG); + const bit = @as(u32, 1) << regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_S; + rst.writeRaw(rst.raw() | bit); + rst.writeRaw(rst.raw() & ~bit); + guard.release(); + + oracle_uart_enable_bus_clock(port, 1); + oracle_uart_sclk_enable(port); + oracle_uart_set_sclk_xtal(port); +} + +// The clock source is selected through oracle_uart_set_sclk_xtal, which names the enumerator on the +// C side. It used to be an integer constant here, and 0 is not XTAL - see that function's comment. + +pub const suite: types.Suite = .{ + .descriptor = .{ + .name = "uart1", + // UART1's block: DR_REG_UART0_BASE + 1 * 0x1000 (soc.h:20). + .base = @intCast(regs.DR_REG_UART0_BASE + 0x1000), + // 40 words, 0x000 through 0x09c. The last register in the block is UART_ID at +0x9c + // (uart_reg.h:1568) and the commit bit UART_REG_UPDATE is at +0x98 - a window that stopped + // at UART_CLK_CONF (+0x88) would be blind to whether the commit even happened, which is the + // single most likely difference against IDF on this peripheral. + .words = 40, + // The read that is a write. See the header. + .no_read = &.{0x00 / 4}, + .volatile_words = &.{ + 0x04 / 4, // UART_INT_RAW - write-1-to-clear, and TXFIFO_EMPTY_INT_RAW moves on its own + 0x08 / 4, // UART_INT_ST - read-only view of the above + 0x1c / 4, // UART_STATUS - live FIFO counts, and the RXD/CTS/DSR pad levels + 0x68 / 4, // UART_MEM_TX_STATUS - FIFO read/write pointers + 0x6c / 4, // UART_MEM_RX_STATUS + 0x70 / 4, // UART_FSM_STATUS - the transmitter's state machine + 0x74 / 4, // UART_POSPULSE - autobaud edge counters, which count whatever the pad does + 0x78 / 4, // UART_NEGPULSE + 0x7c / 4, // UART_LOWPULSE + 0x80 / 4, // UART_HIGHPULSE + 0x84 / 4, // UART_RXD_CNT + 0x90 / 4, // UART_AFIFO_STATUS - the async FIFO's empty/full flags + 0x98 / 4, // UART_REG_UPDATE - self-clearing; reads 0 once the commit lands, but is + // 1 for a few core-clock cycles and a snapshot can catch it + }, + // The APB gate that must read 1 for a snapshot of this block to mean anything. A gated UART + // does not read as zeros, it reads as the last value latched, so two meaningless snapshots + // can compare equal. Pairing from uart_ll.h:257-259, which reads UART1's APB enable out of + // HP_SYS_CLKRST.soc_clk_ctrl2. + .clock = .{ + .reg = @intCast(regs.HP_SYS_CLKRST_SOC_CLK_CTRL2_REG), + .bit = regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_S, + }, + // Reset is the only sound restore for this block: CONF0_SYNC's two FIFO-reset bits and + // REG_UPDATE are write-to-act, and writing a saved word back to offset 0x000 would transmit + // a character. Pairing from uart_ll.h:340-342. Safe here only because this is UART1; + // the same line for UART0 kills the console. + // Reset, and then put the clocking back - which is why this is `.configure` and not + // `.reset_bit`. The harness's reset path does only the pulse, and a UART reset clears the + // core-clock enable and the source select along with everything else. IDF's + // `uart_ll_update` then spins forever waiting for a REG_UPDATE commit that a clockless + // peripheral will never acknowledge: the harness reached the first UART case and stopped, + // with the console silent, looking exactly like a crash. + .restore = .{ .configure = restore }, + }, + .cases = &.{ + // --- baud rate. Four rates spanning the interesting parts of the arithmetic: two ordinary + // ones where the pre-divider is 1, one low enough to need a pre-divider of 33, and one fast + // enough that the integer part gets small and the fraction carries most of the accuracy. + .{ .name = "baudrate", .arg = 115200, .idf = idfBaud115200, .ours = ourBaud115200 }, + .{ .name = "baudrate", .arg = 9600, .idf = idfBaud9600, .ours = ourBaud9600 }, + .{ .name = "baudrate_needs_predivider", .arg = 300, .idf = idfBaud300, .ours = ourBaud300 }, + .{ .name = "baudrate", .arg = 1000000, .idf = idfBaud1M, .ours = ourBaud1M }, + + // --- data format. Each of these is one CONF0_SYNC field plus a commit. + .{ .name = "word_length", .arg = 8, .idf = idfBits8, .ours = ourBits8 }, + .{ .name = "word_length", .arg = 5, .idf = idfBits5, .ours = ourBits5 }, + .{ .name = "stop_bits", .arg = 2, .idf = idfStop2, .ours = ourStop2 }, + .{ .name = "stop_bits_1_5", .arg = 15, .idf = idfStop15, .ours = ourStop15 }, + .{ .name = "parity_odd", .arg = 3, .idf = idfParityOdd, .ours = ourParityOdd }, + .{ .name = "parity_even", .arg = 2, .idf = idfParityEven, .ours = ourParityEven }, + // The asymmetric one: IDF leaves the odd/even bit alone when disabling parity, because 0 + // carries no odd/even information (uart_ll.h:819-822). Setting odd and then disabling is + // the sequence that makes the difference visible, so the case does both. + .{ .name = "parity_odd_then_disable", .idf = idfParityOddThenOff, .ours = ourParityOddThenOff }, + + // --- loopback. Worth a case of its own beyond being one more CONF0_SYNC bit: it is the only + // way to move a byte through this UART with nothing wired to the board. + .{ .name = "loopback_on", .arg = 1, .idf = idfLoopOn, .ours = ourLoopOn }, + .{ .name = "loopback_off", .arg = 0, .idf = idfLoopOff, .ours = ourLoopOff }, + + // --- FIFO resets. These are sequences, not field writes: assert, commit, deassert, commit, + // four stores where a state comparison alone would accept one. Getting the commits wrong + // leaves the register reading exactly as asked and the FIFO not reset. + .{ .name = "txfifo_rst", .idf = idfTxFifoRst, .ours = ourTxFifoRst }, + .{ .name = "rxfifo_rst", .idf = idfRxFifoRst, .ours = ourRxFifoRst }, + + // --- the bare commit, as its own case. If this one differs, every case above is suspect. + .{ .name = "update", .idf = idfUpdate, .ours = ourUpdate }, + + // --- pin routing. Window-blind by construction: the effect is in the GPIO block, so what + // these compare is that neither side disturbs the UART while routing. Kept because a + // routing call that accidentally wrote a UART register would be caught by nothing else, and + // because the pair documents which signal index each side uses. + .{ .name = "route_tx", .arg = 33, .idf = idfRouteTx, .ours = ourRouteTx }, + .{ .name = "route_rx", .arg = 33, .idf = idfRouteRx, .ours = ourRouteRx }, + }, + .setup = setup, +}; + +// ------------------------------------------------------------------------------------- baud rate + +fn idfBaud115200() void { + _ = oracle_uart_set_baudrate(port, 115200, sclk_freq); +} +fn ourBaud115200() void { + _ = ours().setBaudrate(115200, sclk_freq); +} +fn idfBaud9600() void { + _ = oracle_uart_set_baudrate(port, 9600, sclk_freq); +} +fn ourBaud9600() void { + _ = ours().setBaudrate(9600, sclk_freq); +} +fn idfBaud300() void { + _ = oracle_uart_set_baudrate(port, 300, sclk_freq); +} +fn ourBaud300() void { + _ = ours().setBaudrate(300, sclk_freq); +} +fn idfBaud1M() void { + _ = oracle_uart_set_baudrate(port, 1_000_000, sclk_freq); +} +fn ourBaud1M() void { + _ = ours().setBaudrate(1_000_000, sclk_freq); +} + +// ----------------------------------------------------------------------------------- data format +// The numeric arguments to IDF's side are its own enum values from uart_types.h: word length is +// (bits - 5), stop bits are 1/2/3 for 1/1.5/2, parity is 0/2/3 for disable/even/odd. + +fn idfBits8() void { + oracle_uart_set_data_bit_num(port, 3); +} +fn ourBits8() void { + ours().setWordLength(.bits8); +} +fn idfBits5() void { + oracle_uart_set_data_bit_num(port, 0); +} +fn ourBits5() void { + ours().setWordLength(.bits5); +} +fn idfStop2() void { + oracle_uart_set_stop_bits(port, 3); +} +fn ourStop2() void { + ours().setStopBits(.two); +} +fn idfStop15() void { + oracle_uart_set_stop_bits(port, 2); +} +fn ourStop15() void { + ours().setStopBits(.one_and_half); +} +fn idfParityOdd() void { + oracle_uart_set_parity(port, 3); +} +fn ourParityOdd() void { + ours().setParity(.odd); +} +fn idfParityEven() void { + oracle_uart_set_parity(port, 2); +} +fn ourParityEven() void { + ours().setParity(.even); +} +fn idfParityOddThenOff() void { + oracle_uart_set_parity(port, 3); + oracle_uart_set_parity(port, 0); +} +fn ourParityOddThenOff() void { + const u = ours(); + u.setParity(.odd); + u.setParity(.disable); +} + +// -------------------------------------------------------------------------------------- loopback + +fn idfLoopOn() void { + oracle_uart_set_loop_back(port, 1); +} +fn ourLoopOn() void { + ours().setLoopback(true); +} +fn idfLoopOff() void { + oracle_uart_set_loop_back(port, 0); +} +fn ourLoopOff() void { + ours().setLoopback(false); +} + +// ------------------------------------------------------------------------------ FIFO and commit + +fn idfTxFifoRst() void { + oracle_uart_txfifo_rst(port); +} +fn ourTxFifoRst() void { + ours().resetTxFifo(); +} +fn idfRxFifoRst() void { + oracle_uart_rxfifo_rst(port); +} +fn ourRxFifoRst() void { + ours().resetRxFifo(); +} +fn idfUpdate() void { + oracle_uart_update(port); +} +fn ourUpdate() void { + _ = ours().update(); +} + +// ----------------------------------------------------------------------------------- pin routing + +fn idfRouteTx() void { + oracle_uart_route_tx(port, route_pin); +} +fn ourRouteTx() void { + ours().routeTx(route_pin); +} +fn idfRouteRx() void { + oracle_uart_route_rx(port, route_pin); +} +fn ourRouteRx() void { + ours().routeRx(route_pin); +} diff --git a/src/oracle/uart_ref.c b/src/oracle/uart_ref.c new file mode 100644 index 0000000..6476fd4 --- /dev/null +++ b/src/oracle/uart_ref.c @@ -0,0 +1,163 @@ +/* UART's reference implementation: ESP-IDF's own `uart_ll.h`, given external linkage. + * + * No logic here. Anything clever in this file would be a third implementation to doubt, and the + * whole point of the differential is that one side is unmodified IDF. + * + * Two IDF-specific notes: + * + * - Several of these LL functions are shadowed by a macro that references + * `__DECLARE_RCC_ATOMIC_ENV`, an identifier IDF never defines anywhere, so that an unguarded call + * fails to compile: HP_SYS_CLKRST's PERI_CLK_CTRL and SOC_CLK_CTRL registers are shared with + * unrelated peripherals and the only legal caller holds a spinlock. There is no FreeRTOS here and + * core 1 is held in reset at power-on, so declaring the name is as safe as the spinlock would be, + * and it is what IDF's own bootloader does (bootloader_console.c:53). + * - `uart_ll_set_sclk` and `uart_ll_set_baudrate` are *function-like macros* wrapping + * `uart_ll_set_sclk` / `_uart_ll_set_baudrate` (uart_ll.h:477-481, 578-586). Calling the + * underscored inner function directly would skip the very guard above, so these wrappers call + * through the macro. For `set_sclk` the macro and the function share a name, which works only + * because the macro is defined after the function. + */ + +static int __DECLARE_RCC_ATOMIC_ENV __attribute__((unused)); + +#include "hal/uart_ll.h" +#include "soc/uart_struct.h" + +/* Nothing here touches UART0. Resetting UART0 clears UART_CLKDIV, and UART0 is this board's + * console: the output turns to garbage mid-character and the board takes a watchdog reset with + * nothing readable left to say why. Measured on this board. The suite drives UART1. */ +static inline uart_dev_t *dev(unsigned num) +{ + return UART_LL_GET_HW(num); +} + +/* ---------------------------------------------------------------------- clocks, reset, baud */ + +/* Only the operations uart_cases.zig actually pairs are wrapped. A wrapper with no case behind it + * is unreachable code that looks like coverage - and for `uart_ll_reset_register` in particular it + * would be a loaded gun, since the harness restores this block through its own reset-bit path and + * calling that function with num = 0 kills the console. */ + +/* The whole point of the UART suite: IDF's baud-rate arithmetic, which picks an integer pre-divider + * and then a 12-bit divider with a 4-bit fraction, and commits through REG_UPDATE. Returns false on + * the rates it cannot represent - which is a real outcome, not an error path, since a 12-bit divider + * cannot reach every baud from every source clock. uart_ll.h:532-588. */ +int oracle_uart_set_baudrate(unsigned num, unsigned baud, unsigned sclk_freq) +{ + return _uart_ll_set_baudrate(UART_LL_GET_HW(num), baud, sclk_freq) ? 1 : 0; +} + +void oracle_uart_enable_bus_clock(unsigned num, int enable) +{ + uart_ll_enable_bus_clock((uart_port_t)num, enable != 0); +} + +void oracle_uart_sclk_enable(unsigned num) +{ + uart_ll_sclk_enable(dev(num)); +} + + +/* `sel` is the soc_module_clk_t value, not the raw field encoding: UART_SCLK_XTAL, + * UART_SCLK_RTC, UART_SCLK_PLL_F80M. The mapping from those to the 2-bit field is the part of + * uart_ll_set_sclk under test. */ +/* The clock source, named on the C side rather than passed as a number. + * + * `UART_SCLK_XTAL` is an enumerator of `soc_module_clk_t`, not a small ordinal: clk_tree_defs.h:280 + * defines it as SOC_MOD_CLK_XTAL, whose value is whatever position it happens to occupy in a + * chip-wide enum. Passing 0 from Zig - which is what a first version did - lands on an unmatched + * case in `_uart_ll_set_sclk`, whose default is HAL_ASSERT(false); compiled at assertion level 0 + * that is `__builtin_unreachable()`, so the select is left at a value with no clock behind it and + * the very next `uart_ll_update` spins forever on a commit that cannot land. The harness reached the + * first UART case and went silent, which looks exactly like a crash. + * + * Keeping the enumerator on this side of the boundary removes the class of bug entirely. */ +void oracle_uart_set_sclk_xtal(unsigned num) +{ + uart_ll_set_sclk(UART_LL_GET_HW(num), UART_SCLK_XTAL); +} + +void oracle_uart_set_sclk_pll(unsigned num) +{ + uart_ll_set_sclk(UART_LL_GET_HW(num), UART_SCLK_PLL_F80M); +} + +void oracle_uart_set_sclk(unsigned num, unsigned sel) +{ + uart_ll_set_sclk(dev(num), (soc_module_clk_t)sel); +} + + +/* -------------------------------------------------------------------------------- data format */ + +void oracle_uart_set_data_bit_num(unsigned num, unsigned bits) +{ + uart_ll_set_data_bit_num(dev(num), (uart_word_length_t)bits); +} + +void oracle_uart_set_stop_bits(unsigned num, unsigned stop) +{ + uart_ll_set_stop_bits(dev(num), (uart_stop_bits_t)stop); +} + +void oracle_uart_set_parity(unsigned num, unsigned parity) +{ + uart_ll_set_parity(dev(num), (uart_parity_t)parity); +} + +/* ---------------------------------------------------------------------------------------- FIFO */ + +void oracle_uart_txfifo_rst(unsigned num) +{ + uart_ll_txfifo_rst(dev(num)); +} + +void oracle_uart_rxfifo_rst(unsigned num) +{ + uart_ll_rxfifo_rst(dev(num)); +} + + +/* ------------------------------------------------------------------------- loopback and update */ + +void oracle_uart_set_loop_back(unsigned num, int enable) +{ + uart_ll_set_loop_back(dev(num), enable != 0); +} + +void oracle_uart_update(unsigned num) +{ + uart_ll_update(dev(num)); +} + +/* ---------------------------------------------------------------------------------- pin routing */ + +/* IDF routes UART pins through esp_rom_gpio_connect_*_signal / gpio_ll, not through uart_ll, so the + * reference for pin routing is the GPIO LL - the same functions gpio_ref.c wraps, called with this + * UART's signal indices. Kept here rather than in gpio_ref.c because the *signal index* is the part + * under test, and it belongs to the UART. */ +#include "hal/gpio_ll.h" +#include "soc/gpio_struct.h" +/* The signal indices themselves, which are not in any *_ll.h - and are the whole point of these two + * wrappers: the Zig side reads the same macros through translate-c. */ +#include "soc/gpio_sig_map.h" + +void oracle_uart_route_tx(unsigned num, unsigned pin) +{ + const unsigned sig[5] = { + UART0_TXD_PAD_OUT_IDX, UART1_TXD_PAD_OUT_IDX, UART2_TXD_PAD_OUT_IDX, + UART3_TXD_PAD_OUT_IDX, UART4_TXD_PAD_OUT_IDX, + }; + gpio_ll_set_output_signal_matrix_source(&GPIO, pin, sig[num], false); + gpio_ll_set_output_enable_ctrl(&GPIO, pin, true, false); +} + +void oracle_uart_route_rx(unsigned num, unsigned pin) +{ + const unsigned sig[5] = { + UART0_RXD_PAD_IN_IDX, UART1_RXD_PAD_IN_IDX, UART2_RXD_PAD_IN_IDX, + UART3_RXD_PAD_IN_IDX, UART4_RXD_PAD_IN_IDX, + }; + gpio_ll_input_enable(&GPIO, pin); + gpio_ll_set_input_signal_matrix_source(&GPIO, sig[num], pin, false); +} diff --git a/src/soc.zig b/src/soc.zig new file mode 100644 index 0000000..f022b50 --- /dev/null +++ b/src/soc.zig @@ -0,0 +1,132 @@ +//! ESP32-P4 peripherals, modelled at comptime. +//! +//! There is no HAL here and no generated 20k-line register header: a `Reg` is a typed pointer to +//! an MMIO word, and a peripheral is a struct of them. Everything is `inline`, so `gpio.setHigh(20)` +//! compiles to the single `sw` instruction it should be, and a wrong bit index is a compile error +//! rather than a silent write. +//! +//! Addresses are from ESP-IDF v6.0.2 `components/soc/esp32p4/register/hw_ver1/soc/` - the pre-v3 +//! header set, which is the one that matches this silicon (rev v1.3) - except the GPIO matrix +//! signal index, which lives in `components/soc/esp32p4/include/soc/gpio_sig_map.h`. + +const std = @import("std"); + +/// A 32-bit memory-mapped register. +pub fn Reg(comptime addr: usize) type { + return struct { + pub const address = addr; + const ptr: *volatile u32 = @ptrFromInt(addr); + + pub inline fn read() u32 { + return ptr.*; + } + pub inline fn write(value: u32) void { + ptr.* = value; + } + pub inline fn set(mask: u32) void { + ptr.* = ptr.* | mask; + } + pub inline fn clear(mask: u32) void { + ptr.* = ptr.* & ~mask; + } + /// Read-modify-write a bitfield: `modify(.{ .shift = 12, .width = 3 }, 5)`. + pub inline fn modify(comptime field: Field, value: u32) void { + const mask: u32 = ((@as(u32, 1) << field.width) - 1) << field.shift; + ptr.* = (ptr.* & ~mask) | ((value << field.shift) & mask); + } + }; +} + +pub const Field = struct { shift: u5, width: u5 }; + +/// A register with a named layout: pass a packed struct whose bit width is 32 and the accessors +/// become typed, so a pad is configured by naming fields instead of shifting bits. Read-modify- +/// write stays explicit - `var v = reg.read(); v.mcu_sel = 1; reg.write(v);` - because that is one +/// load and one store, and hiding it behind a partial-update type buys nothing here. +pub fn Typed(comptime T: type, comptime addr: usize) type { + comptime std.debug.assert(@bitSizeOf(T) == 32); + return struct { + pub const address = addr; + const ptr: *volatile T = @ptrFromInt(addr); + + pub inline fn read() T { + return ptr.*; + } + pub inline fn write(value: T) void { + ptr.* = value; + } + /// Apply `f` to the current value and write the result back. + pub inline fn modify(comptime f: fn (T) T) void { + ptr.* = f(ptr.*); + } + }; +} + +/// An array of identical registers. The index type is narrowed to the array's real range, so an +/// out-of-range access is a compile error in every optimize mode - an `assert` would have been +/// compiled out under ReleaseSmall, which is this project's default. +pub fn RegArray(comptime base: usize, comptime stride: usize, comptime count: usize) type { + return struct { + pub const Index = std.math.IntFittingRange(0, count - 1); + + /// comptime, because `IntFittingRange` rounds up to a whole width: for a 57-entry array the + /// index type is u6, which would happily accept 57..63. Every caller here passes a comptime + /// pin anyway, so this costs nothing and makes the bound real in all optimize modes. + pub inline fn at(comptime index: Index) *volatile u32 { + comptime std.debug.assert(index < count); + return @ptrFromInt(base + @as(usize, index) * stride); + } + }; +} + +const hp_periph1 = 0x500C0000; + +/// GPIO and the IO MUX now live in the HAL, which builds them out of ESP-IDF's own register macros +/// (`hal/gpio.zig`) instead of the hand-transcribed addresses that used to be here. The +/// transcription is exactly the kind of thing that goes quietly wrong: this file's matrix constant +/// said 256 with a comment warning that the S3's is 128, and the first hand-written replacement in +/// the HAL used 128 anyway. It now comes from `SIG_GPIO_OUT_IDX` in IDF's `gpio_sig_map.h`. +pub const gpio = @import("hal").gpio; + +/// Mask ROM routines. These are the only "library" a bare image links against: the addresses come +/// from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld` and the linker script re-declares them. +pub const rom = struct { + pub extern fn ets_printf(fmt: [*:0]const u8, ...) c_int; + pub extern fn ets_delay_us(us: u32) void; + + pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void { + _ = @call(.auto, ets_printf, .{fmt} ++ args); + } +}; + +/// Busy-wait for a number of CPU cycles, using the cycle counter rather than the mask ROM. Useful +/// when an image must not depend on ROM entry points at all, and for delays shorter than the ROM's +/// microsecond granularity. +pub inline fn delayCycles(n: u64) void { + const start = cycles(); + while (cycles() - start < n) {} +} + +/// Cycle counter: CSR 0xC00/0xC80, i.e. `cycle`/`cycleh` - the unprivileged shadows of mcycle, and +/// what ESP-IDF itself reads on this part (`rv_utils.h`: `RV_READ_CSR(cycle)`, because +/// SOC_CPU_HAS_CSR_PC is not defined for the P4). +/// +/// Read high-low-high: two separate CSR reads can straddle a wrap of the low word, which would +/// otherwise report a value 2^32 too large roughly every 47 seconds at 90 MHz. +pub inline fn cycles() u64 { + while (true) { + var hi0: u32 = undefined; + var lo: u32 = undefined; + var hi1: u32 = undefined; + asm volatile ("csrr %[r], 0xC80" + : [r] "=r" (hi0), + ); + asm volatile ("csrr %[r], 0xC00" + : [r] "=r" (lo), + ); + asm volatile ("csrr %[r], 0xC80" + : [r] "=r" (hi1), + ); + if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo; + } +} |
