summaryrefslogtreecommitdiff
path: root/src/hal
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-25 12:40:53 -0300
committerGabriel Schneider <[email protected]>2026-08-25 12:46:51 -0300
commitf5f8068fac59b4f16046c2022c2fc7c7e447ef4c (patch)
tree2731a3ed4e51cae09e184e25778eded5fc37d1f5 /src/hal
downloadesp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.tar.gz
esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.zip
zig-p4: pure-Zig ESP32-P4 toolchain
build.zig generates the linker script and drives Zig's own LLD; tools/image.zig turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker. src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers; src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip; src/oracle/ diffs this HAL against ESP-IDF's on the die.
Diffstat (limited to 'src/hal')
-rw-r--r--src/hal/clkrst.zig265
-rw-r--r--src/hal/gpio.zig471
-rw-r--r--src/hal/i2c.zig1091
-rw-r--r--src/hal/intr.zig965
-rw-r--r--src/hal/ledc.zig587
-rw-r--r--src/hal/rwdt.zig123
-rw-r--r--src/hal/sdmmc.zig2002
-rw-r--r--src/hal/systimer.zig151
-rw-r--r--src/hal/timg.zig513
-rw-r--r--src/hal/uart.zig622
10 files changed, 6790 insertions, 0 deletions
diff --git a/src/hal/clkrst.zig b/src/hal/clkrst.zig
new file mode 100644
index 0000000..89a27ef
--- /dev/null
+++ b/src/hal/clkrst.zig
@@ -0,0 +1,265 @@
+//! Peripheral clock gates and resets: HP_SYS_CLKRST.
+//!
+//! Two things about this block are counter-intuitive on the ESP32-P4, and both were found by
+//! reading ESP-IDF rather than by assuming:
+//!
+//! **Peripheral clocks are already on.** `esp_system/port/soc/esp32p4/clk.c:200` says so in as many
+//! words - "All peripheral clocks are default enabled after chip is powered on" - and the reset
+//! values in `hp_sys_clkrst_reg.h` agree: REG_UART0_APB_CLK_EN, REG_TIMERGRP0_APB_CLK_EN,
+//! REG_SYSTIMER_APB_CLK_EN and REG_IOMUX_APB_CLK_EN all default to 1, with their RST_EN bits at 0.
+//! An image that boots from the stock second-stage bootloader never runs `esp_perip_clk_init`, so it
+//! inherits those defaults. So this file is not a prerequisite for touching a peripheral; it is what
+//! you need to *re*-initialise one, and to reach the few blocks that really are gated off (TWAI is
+//! the notable one: REG_TWAI0_APB_CLK_EN defaults to 0).
+//!
+//! **The hazard is atomicity, not gating.** Every gate and reset bit for the whole chip lives in a
+//! handful of shared registers, so `enable(.uart0)` is a read-modify-write of a word that also holds
+//! the gate for unrelated peripherals. ESP-IDF makes unguarded calls impossible to compile by
+//! referencing `__DECLARE_RCC_ATOMIC_ENV`, an identifier it never defines anywhere; the only legal
+//! callers are inside `PERIPH_RCC_ATOMIC()`, which takes a FreeRTOS spinlock. There is no FreeRTOS
+//! here and core 1 is held in reset at power-on (`hp_sys_clkrst_reg.h`: REG_RST_EN_CORE1_GLOBAL
+//! defaults to 1), so masking interrupts around the read-modify-write is sufficient and is what
+//! `atomically` does.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+// The four shared registers this file touches. Which field lives in which register is not derivable
+// from the macro names - `HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_S` does not say `SOC_CLK_CTRL2` - so the
+// pairing is taken from ESP-IDF's own LL, cited per peripheral below.
+const soc_clk_ctrl1 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG);
+const soc_clk_ctrl2 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL2_REG);
+const soc_clk_ctrl3 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL3_REG);
+/// SDMMC's reset bit is not in HP_SYS_CLKRST at all. `sdmmc_ll_reset_register`
+/// (`sdmmc_ll.h:158-163`) writes `LP_AON_CLKRST.hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc`, a register
+/// of the *low-power* always-on clock-and-reset block, which it shares with the Ethernet MAC. So
+/// the `Gates.reset` field is a register as well as a bit, and this is the row that proves it has
+/// to be.
+const lp_hp_sdmmc_emac_rst_ctrl = Reg.at(regs.LP_CLKRST_HP_SDMMC_EMAC_RST_CTRL_REG);
+const hp_rst_en1 = Reg.at(regs.HP_SYS_CLKRST_HP_RST_EN1_REG);
+
+/// Interrupts masked for the duration of a read-modify-write on a shared register:
+///
+/// const guard = clkrst.maskInterrupts();
+/// defer guard.release();
+///
+/// mstatus.MIE is bit 3. `csrrc` clears it and returns the previous mstatus in one instruction, and
+/// `release` restores only what was actually there - so this composes: using it inside code that
+/// already had interrupts off does not turn them on at the end.
+pub const Guard = struct {
+ prev_mie: bool,
+
+ pub inline fn release(self: Guard) void {
+ if (self.prev_mie) {
+ asm volatile ("csrs mstatus, %[mask]"
+ :
+ : [mask] "r" (@as(u32, 1 << 3)),
+ );
+ }
+ }
+};
+
+pub inline fn maskInterrupts() Guard {
+ const prev = asm volatile ("csrrc %[out], mstatus, %[mask]"
+ : [out] "=r" (-> u32),
+ : [mask] "r" (@as(u32, 1 << 3)),
+ );
+ return .{ .prev_mie = prev & (1 << 3) != 0 };
+}
+
+/// A peripheral's clock gates and reset bit.
+///
+/// `sys_clk` is present only where the peripheral has a second gate on the SYS clock as well as the
+/// APB one; UART has both (uart_ll.h:252-253 reads `soc_clk_ctrl2.reg_uart0_apb_clk_en` and
+/// `soc_clk_ctrl1.reg_uart0_sys_clk_en`), most blocks have only APB.
+const Gates = struct {
+ apb_clk: ?struct { reg: Reg, field: Field } = null,
+ sys_clk: ?struct { reg: Reg, field: Field } = null,
+ reset: struct { reg: Reg, field: Field },
+ /// TIMG only: resetting the block re-arms flash-boot protection, which reboots the board a
+ /// moment later with no diagnostic. `timg_ll.h:53-72` documents it and clears the bit as part of
+ /// the reset; anything that resets TIMG must do the same.
+ clears_flashboot: bool = false,
+};
+
+pub const Peripheral = enum {
+ uart0,
+ uart1,
+ uart2,
+ uart3,
+ uart4,
+ timg0,
+ timg1,
+ systimer,
+ twai0,
+ ledc,
+ i2c0,
+ i2c1,
+ sdmmc,
+
+ fn gates(comptime self: Peripheral) Gates {
+ return switch (self) {
+ // uart_ll.h:251-253 for UART0, and the same three fields per instance after it.
+ .uart0 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_V) },
+ .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_SYS_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART0_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART0_APB_V) },
+ },
+ .uart1 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_V) },
+ .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART1_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART1_SYS_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_V) },
+ },
+ .uart2 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART2_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART2_APB_CLK_EN_V) },
+ .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART2_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART2_SYS_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART2_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART2_APB_V) },
+ },
+ .uart3 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART3_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART3_APB_CLK_EN_V) },
+ .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART3_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART3_SYS_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART3_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART3_APB_V) },
+ },
+ .uart4 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART4_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART4_APB_CLK_EN_V) },
+ .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART4_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART4_SYS_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART4_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART4_APB_V) },
+ },
+ // timg_ll.h:35-42 for the gate, :60-72 for the reset. The timer groups' APB gate is in
+ // SOC_CLK_CTRL2 - the same word as the UARTs' - not in PERI_CLK_CTRL21. An earlier
+ // version of this table had these four entries in PERI_CLK_CTRL21 and so wrote bits
+ // 21-24 of an unrelated register; hp_sys_clkrst_reg.h:605 defines SOC_CLK_CTRL2_REG and
+ // :753/:763/:770/:777 put TIMERGRP0 at bit 21, TIMERGRP1 at 22, SYSTIMER at 23 and
+ // TWAI0 at 24 inside it. PERI_CLK_CTRL20/21 do hold timer-group fields - the per-timer
+ // clock source and gate, see hal/timg.zig - which is what made the mix-up plausible.
+ //
+ // It survived a hardware check because `isClockEnabled` read back the same wrong bit
+ // `setClockEnabled` had just written: self-consistent, and independent of the chip.
+ .timg0 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP0_S, regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP0_V) },
+ .clears_flashboot = true,
+ },
+ .timg1 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP1_S, regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP1_V) },
+ .clears_flashboot = true,
+ },
+ // systimer_ll.h:71-72.
+ .systimer => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_SYSTIMER_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SYSTIMER_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_STIMER_S, regs.HP_SYS_CLKRST_REG_RST_EN_STIMER_V) },
+ },
+ // The one block whose clock is gated OFF at power-on, which makes it the only peripheral
+ // where `enable` is observably necessary rather than merely correct.
+ .twai0 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TWAI0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TWAI0_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TWAI0_S, regs.HP_SYS_CLKRST_REG_RST_EN_TWAI0_V) },
+ },
+ // ledc_ll.h:135 for the gate (`HP_SYS_CLKRST.soc_clk_ctrl3.reg_ledc_apb_clk_en`) and
+ // :150 for the reset (`hp_rst_en1.reg_rst_en_ledc`). LEDC's APB gate is the *first* bit
+ // of SOC_CLK_CTRL3, a third register this table did not previously need, and it is one
+ // of the few whose reset value is 0 (hp_sys_clkrst_reg.h:835): LEDC's registers are
+ // gated off at power-on, so `setClockEnabled(.ledc, true)` is a prerequisite and not a
+ // formality. LEDC's *function* clock and its source mux live in PERI_CLK_CTRL22
+ // (ledc_ll.h:179, :241) and belong to the peripheral, not to this table - see
+ // hal/ledc.zig.
+ .ledc => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl3, .field = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_S, regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_V) },
+ },
+ // i2c_ll.h:149-156 for the gates (`HP_SYS_CLKRST.soc_clk_ctrl2.reg_i2c0_apb_clk_en`,
+ // and `reg_i2c1_apb_clk_en` for port 1) and :167-176 for the resets
+ // (`hp_rst_en1.reg_rst_en_i2c0` / `_i2c1`). Both APB gates default to 1
+ // (hp_sys_clkrst_reg.h:694-703), so the registers are reachable from boot; what I2C
+ // does *not* get from this table is its controller clock, whose enable, source mux and
+ // divider are I2C-specific fields of PERI_CLK_CTRL10/11 and live in hal/i2c.zig. That
+ // one defaults to 0, so an I2C port brought up through this table alone has readable
+ // registers and a state machine that never moves.
+ .i2c0 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_S, regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_V) },
+ },
+ .i2c1 => .{
+ .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_APB_CLK_EN_V) },
+ .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_I2C1_S, regs.HP_SYS_CLKRST_REG_RST_EN_I2C1_V) },
+ },
+ // The one row in this table whose two halves live in two different peripherals, and
+ // the one whose clock really is gated off at power-on alongside LEDC's.
+ //
+ // `sdmmc_ll.h:140-144` is the gate: `HP_SYS_CLKRST.soc_clk_ctrl1.reg_sdmmc_sys_clk_en`,
+ // a *SYS* clock and not an APB one - SDMMC has no APB gate at all, which is why the
+ // `apb_clk` field is absent here rather than merely unused. It defaults to 0
+ // (hp_sys_clkrst_reg.h:475-481, "default: 0"), so `setClockEnabled(.sdmmc, true)` is a
+ // prerequisite for the register block reading anything but stale values.
+ //
+ // `sdmmc_ll.h:158-163` is the reset, and it is in LP_AON_CLKRST:
+ // `hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc`, bit 28 (lp_clkrst_reg.h:993-999). Looking for
+ // an `HP_SYS_CLKRST_REG_RST_EN_SDMMC` finds nothing, which is exactly the shape of the
+ // mistake the timer-group rows above record: a plausible name in the wrong register.
+ //
+ // The host clock generator - source mux, divider, sampling phase - is *not* here. It
+ // is SDMMC-specific and lives in PERI_CLK_CTRL01/02, in hal/sdmmc.zig, the same
+ // division this table makes for I2C and LEDC.
+ .sdmmc => .{
+ .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_V) },
+ .reset = .{ .reg = lp_hp_sdmmc_emac_rst_ctrl, .field = Field.of(regs.LP_CLKRST_RST_EN_SDMMC_S, regs.LP_CLKRST_RST_EN_SDMMC_V) },
+ },
+ };
+ }
+};
+
+/// Turn a peripheral's bus clocks on or off.
+pub fn setClockEnabled(comptime p: Peripheral, on: bool) void {
+ const g = comptime p.gates();
+ const v: u32 = @intFromBool(on);
+ const guard = maskInterrupts();
+ defer guard.release();
+ if (g.sys_clk) |s| s.reg.modify(.{s.field.is(v)});
+ if (g.apb_clk) |a| a.reg.modify(.{a.field.is(v)});
+}
+
+/// Whether the peripheral's bus clock is on.
+///
+/// APB gate if it has one, SYS gate otherwise: SDMMC has only the latter (`sdmmc_ll.h:140-144`),
+/// and answering `true` unconditionally for it would have made the oracle's clock check - the one
+/// that exists because a gated block reads stale rather than zero - pass on a gated block.
+pub fn isClockEnabled(comptime p: Peripheral) bool {
+ const g = comptime p.gates();
+ if (g.apb_clk) |a| return a.reg.get(a.field) == 1;
+ if (g.sys_clk) |s| return s.reg.get(s.field) == 1;
+ return true;
+}
+
+/// Pulse a peripheral's reset: assert, deassert.
+///
+/// For the timer groups this also clears flash-boot watchdog protection, which the reset re-arms.
+/// Leaving that out reboots the board a moment later with nothing on the console to explain it.
+pub fn resetPeripheral(comptime p: Peripheral) void {
+ const g = comptime p.gates();
+ {
+ const guard = maskInterrupts();
+ defer guard.release();
+ g.reset.reg.modify(.{g.reset.field.is(1)});
+ g.reset.reg.modify(.{g.reset.field.is(0)});
+ }
+ if (comptime g.clears_flashboot) {
+ const wdtconfig0 = Reg.atAddress(switch (p) {
+ .timg0 => regs.TIMG_WDTCONFIG0_REG(0),
+ .timg1 => regs.TIMG_WDTCONFIG0_REG(1),
+ else => unreachable,
+ });
+ wdtconfig0.modify(.{Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V).is(0)});
+ }
+}
+
+/// Reset a peripheral and make sure its clocks are on, in that order: a peripheral configured
+/// before its reset is released loses the configuration.
+pub fn init(comptime p: Peripheral) void {
+ setClockEnabled(p, true);
+ resetPeripheral(p);
+}
diff --git a/src/hal/gpio.zig b/src/hal/gpio.zig
new file mode 100644
index 0000000..88a8675
--- /dev/null
+++ b/src/hal/gpio.zig
@@ -0,0 +1,471 @@
+//! GPIO and the IO MUX.
+//!
+//! The P4 has 57 pins (GPIO0-56) and every whole-bank register is therefore split in two: `out`
+//! covers 0-31 and `out1` covers 32-56. Getting that split wrong is the classic P4 GPIO bug - a
+//! write to `out` with a shift of 40 lands on pin 8 - so the bank arithmetic lives in exactly one
+//! place here (`Bank`) and every operation goes through it.
+//!
+//! Levels and enables are driven through the `_W1TS`/`_W1TC` (write-1-to-set / write-1-to-clear)
+//! aliases rather than read-modify-write on `out`/`enable`. That is what ESP-IDF's LL does, and it
+//! is not a style choice: a read-modify-write of a whole bank races anything else touching another
+//! pin in the same bank, and there is no lock here to prevent it.
+//!
+//! Pad configuration (direction of the *input* buffer, pulls, drive strength, function select) is
+//! not in the GPIO peripheral at all - it is in the IO MUX, one register per pad. The two must be
+//! kept in step: a pin driven by `enable` but with `fun_ie` clear cannot be read back, which is the
+//! single most common "my GPIO does not work" on this part.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+/// GPIO0-56. 57 pins, and the last five (52-56) exist only on some packages.
+pub const max_pin = 56;
+pub const pin_count = max_pin + 1;
+
+/// Which half of a split bank register a pin lives in, and its bit inside that half.
+const Bank = struct {
+ high: bool,
+ bit: u5,
+
+ inline fn of(pin: u8) Bank {
+ std.debug.assert(pin <= max_pin);
+ return if (pin < 32)
+ .{ .high = false, .bit = @intCast(pin) }
+ else
+ .{ .high = true, .bit = @intCast(pin - 32) };
+ }
+
+ inline fn mask(self: Bank) u32 {
+ return @as(u32, 1) << self.bit;
+ }
+
+ inline fn pick(self: Bank, lo: Reg, hi: Reg) Reg {
+ return if (self.high) hi else lo;
+ }
+};
+
+// The whole-bank registers. `_W1TS`/`_W1TC` are separate addresses that set or clear only the bits
+// written as 1, which is what makes a single-pin update atomic against the rest of the bank.
+const out = Reg.at(regs.GPIO_OUT_REG);
+const out1 = Reg.at(regs.GPIO_OUT1_REG);
+const out_w1ts = Reg.at(regs.GPIO_OUT_W1TS_REG);
+const out1_w1ts = Reg.at(regs.GPIO_OUT1_W1TS_REG);
+const out_w1tc = Reg.at(regs.GPIO_OUT_W1TC_REG);
+const out1_w1tc = Reg.at(regs.GPIO_OUT1_W1TC_REG);
+const enable_w1ts = Reg.at(regs.GPIO_ENABLE_W1TS_REG);
+const enable1_w1ts = Reg.at(regs.GPIO_ENABLE1_W1TS_REG);
+const enable_w1tc = Reg.at(regs.GPIO_ENABLE_W1TC_REG);
+const enable1_w1tc = Reg.at(regs.GPIO_ENABLE1_W1TC_REG);
+const enable = Reg.at(regs.GPIO_ENABLE_REG);
+const enable1 = Reg.at(regs.GPIO_ENABLE1_REG);
+const in = Reg.at(regs.GPIO_IN_REG);
+const in1 = Reg.at(regs.GPIO_IN1_REG);
+
+/// One IO MUX register per pad, stride taken from two consecutive macros rather than assumed.
+const pad = mmio.RegArray(
+ regs.PERIPHS_IO_MUX_U_PAD_GPIO0,
+ regs.PERIPHS_IO_MUX_U_PAD_GPIO1,
+ pin_count,
+);
+
+// Pad fields. These macros are unprefixed globals in io_mux_reg.h - they describe every pad, not
+// one - which is why they read as bare `MCU_SEL` rather than `IO_MUX_GPIO7_MCU_SEL`.
+const fun_ie = Field.of(regs.FUN_IE_S, regs.FUN_IE_V);
+const fun_drv = Field.of(regs.FUN_DRV_S, regs.FUN_DRV_V);
+const mcu_sel = Field.of(regs.MCU_SEL_S, regs.MCU_SEL_V);
+// io_mux_reg.h defines no macros for the two pull bits; io_mux_struct.h documents them as
+// `fun_wpd : R/W; bitpos: [7]` and `fun_wpu : R/W; bitpos: [8]`.
+const fun_wpd = Field.bit(7);
+const fun_wpu = Field.bit(8);
+
+/// IO MUX function for a pad. Function 1 is plain GPIO on every P4 pad; the others select a
+/// peripheral wired directly to that pad, and anything not on this list has to go through the GPIO
+/// matrix instead.
+pub const Function = enum(u3) {
+ f0 = 0,
+ /// Plain GPIO - the GPIO peripheral drives and samples the pad.
+ gpio = 1,
+ f2 = 2,
+ f3 = 3,
+ f4 = 4,
+ f5 = 5,
+ f6 = 6,
+ f7 = 7,
+};
+
+pub const Drive = enum(u2) {
+ /// ~5 mA
+ weakest = 0,
+ /// ~10 mA
+ weak = 1,
+ /// ~20 mA, the reset value
+ medium = 2,
+ /// ~40 mA
+ strong = 3,
+};
+
+pub const Pull = enum { none, up, down };
+
+// ------------------------------------------------------------------------------------- levels
+
+/// Drive a pin high or low. Uses the write-1-to-set/clear alias, so no other pin in the bank is
+/// disturbed and no read is needed.
+pub inline fn setLevel(pin: u8, level: u1) void {
+ const b = Bank.of(pin);
+ const r = if (level == 1)
+ b.pick(out_w1ts, out1_w1ts)
+ else
+ b.pick(out_w1tc, out1_w1tc);
+ r.writeRaw(b.mask());
+}
+
+pub inline fn setHigh(pin: u8) void {
+ setLevel(pin, 1);
+}
+
+pub inline fn setLow(pin: u8) void {
+ setLevel(pin, 0);
+}
+
+pub inline fn toggle(pin: u8) void {
+ const b = Bank.of(pin);
+ if (b.pick(out, out1).raw() & b.mask() != 0) setLow(pin) else setHigh(pin);
+}
+
+/// Sample the pad. Reads the *input* register, so it reports what the pin is actually at - which
+/// for an open-drain or externally driven pin is not necessarily what was last written to `out`.
+/// Requires the pad's input buffer to be enabled (`setInputEnable`).
+pub inline fn getLevel(pin: u8) u1 {
+ const b = Bank.of(pin);
+ return @intCast((b.pick(in, in1).raw() >> b.bit) & 1);
+}
+
+/// What was last driven, from the output register rather than the pad.
+pub inline fn getDrivenLevel(pin: u8) u1 {
+ const b = Bank.of(pin);
+ return @intCast((b.pick(out, out1).raw() >> b.bit) & 1);
+}
+
+// -------------------------------------------------------------------------------- direction
+
+pub inline fn outputEnable(pin: u8) void {
+ const b = Bank.of(pin);
+ b.pick(enable_w1ts, enable1_w1ts).writeRaw(b.mask());
+}
+
+pub inline fn outputDisable(pin: u8) void {
+ const b = Bank.of(pin);
+ b.pick(enable_w1tc, enable1_w1tc).writeRaw(b.mask());
+}
+
+pub inline fn isOutputEnabled(pin: u8) bool {
+ const b = Bank.of(pin);
+ return b.pick(enable, enable1).raw() & b.mask() != 0;
+}
+
+/// The pad's input buffer. Independent of the output driver: both can be on at once, which is how a
+/// pin is read back while being driven.
+pub inline fn setInputEnable(pin: u8, on: bool) void {
+ pad.at(pin).modify(.{fun_ie.is(@intFromBool(on))});
+}
+
+/// Whether the pad's input buffer is on. The counterpart of `setInputEnable`, and worth having
+/// because a routed input with `fun_ie` clear is indistinguishable from a card that never drove
+/// the pin: both read as a constant.
+pub inline fn isInputEnabled(pin: u8) bool {
+ return pad.at(pin).get(fun_ie) != 0;
+}
+
+// -------------------------------------------------------------------------------- pad config
+
+pub inline fn setFunction(pin: u8, f: Function) void {
+ pad.at(pin).modify(.{mcu_sel.is(@intFromEnum(f))});
+}
+
+pub inline fn setDrive(pin: u8, d: Drive) void {
+ pad.at(pin).modify(.{fun_drv.is(@intFromEnum(d))});
+}
+
+/// Internal pull resistors. Setting one direction always clears the other in the same store: a pad
+/// with both enabled is a fight between two resistors, and it is easy to reach by two calls.
+pub inline fn setPull(pin: u8, p: Pull) void {
+ pad.at(pin).modify(.{
+ fun_wpu.is(@intFromBool(p == .up)),
+ fun_wpd.is(@intFromBool(p == .down)),
+ });
+}
+
+/// What `setPull` last left, read back from the pad. A pad with both resistors enabled cannot be
+/// reached through `setPull`, but the reset value or another driver can leave one that way, so the
+/// contradictory case is reported as `.none` rather than picking a winner.
+pub inline fn getPull(pin: u8) Pull {
+ const w = pad.at(pin).raw();
+ const up = w & fun_wpu.mask() != 0;
+ const down = w & fun_wpd.mask() != 0;
+ if (up and !down) return .up;
+ if (down and !up) return .down;
+ return .none;
+}
+
+/// Open-drain: the pad drives low and releases high instead of driving both rails.
+///
+/// This one is not in the IO MUX with the other pad properties - it is `GPIO_PINn_PAD_DRIVER`, bit
+/// 2 of the GPIO peripheral's per-pin register (`gpio_reg.h:363-368`, "1:open-drain. 0:normal"),
+/// which is a different register file from `PERIPHS_IO_MUX_U_PAD_GPIOn`. A shared bus - I2C, or any
+/// wired-AND signal - needs this on both pads *and* an external pull-up; the internal pull-up is
+/// too weak for anything but a short trace at a low bit rate.
+pub inline fn setOpenDrain(pin: u8, on: bool) void {
+ pin_cfg.at(pin).modify(.{pad_driver.is(@intFromBool(on))});
+}
+
+/// The GPIO peripheral's per-pin configuration register, one per pad. Not the IO MUX: this file
+/// holds the open-drain select, the interrupt configuration and the input synchroniser bypasses.
+const pin_cfg = mmio.RegArray(regs.GPIO_PIN0_REG, regs.GPIO_PIN1_REG, pin_count);
+const pad_driver = Field.of(regs.GPIO_PIN0_PAD_DRIVER_S, regs.GPIO_PIN0_PAD_DRIVER_V);
+
+// -------------------------------------------------------------------------- pin interrupts
+
+/// How a pad raises its interrupt. `gpio_reg.h:377-381`: "0:disable GPIO interrupt. 1:trigger at
+/// posedge. 2:trigger at negedge. 3:trigger at any edge. 4:valid at low level. 5:valid at high
+/// level".
+pub const IntrType = enum(u3) {
+ disable = 0,
+ posedge = 1,
+ negedge = 2,
+ anyedge = 3,
+ low_level = 4,
+ high_level = 5,
+};
+
+const int_type = Field.of(regs.GPIO_PIN0_INT_TYPE_S, regs.GPIO_PIN0_INT_TYPE_V);
+/// Five bits, one per consumer of the pad's interrupt, not a boolean. `gpio_reg.h:400-402` says
+/// "set bit 13 to enable CPU interrupt, set bit 14 to enable CPU(not shielded) interrupt", and
+/// `gpio_ll.h:41,213` names bit 0 of the field `GPIO_LL_INTR0_ENA` and writes exactly that to
+/// route a pad to the `gpio_intr0` source. Writing 1 here means "line 0", not "enabled".
+const int_ena = Field.of(regs.GPIO_PIN0_INT_ENA_S, regs.GPIO_PIN0_INT_ENA_V);
+
+/// Which of the P4's four GPIO interrupt outputs a pad drives. Each is a separate entry in the
+/// interrupt matrix (`hal.intr.Source.gpio_intr0` .. `gpio_intr3`), and each has its own status
+/// register pair. ESP-IDF only ever uses line 0 - `gpio_ll_intr_enable_on_core` hard-codes
+/// `GPIO_LL_INTR0_ENA` with a "TODO: IDF-7995" beside it - so line 0 is the tested path.
+pub const IntrLine = enum(u3) {
+ line0 = 0,
+ line1 = 1,
+ line2 = 2,
+ line3 = 3,
+};
+
+/// Per-line status, gated by `int_ena`. Reading `status`/`status1` instead would report pads whose
+/// interrupt is configured but routed to a different line. `gpio_reg.h:277,291` for line 0,
+/// `:302,316` for line 1; lines 2 and 3 continue the same +0x8 stride.
+const intr_status = mmio.RegArray(regs.GPIO_INTR_0_REG, regs.GPIO_INTR_1_REG, 4);
+const intr_status1 = mmio.RegArray(regs.GPIO_INTR1_0_REG, regs.GPIO_INTR1_1_REG, 4);
+
+/// Status is cleared through a shared write-1-to-clear register, not a per-line one: one pad has
+/// one latch however many lines observe it. `gpio_reg.h:233,269`.
+const status_w1tc = Reg.at(regs.GPIO_STATUS_W1TC_REG);
+const status1_w1tc = Reg.at(regs.GPIO_STATUS1_W1TC_REG);
+
+/// Arm a pad's interrupt and route it to one of the four GPIO interrupt outputs.
+///
+/// This is the GPIO peripheral's half only. The other half is `hal.intr`: the chosen line still
+/// has to be routed from `Source.gpio_intr0`+n to a CLIC line and given a handler. Doing it in two
+/// calls is deliberate - one pad's interrupt and one CPU line are not the same resource, and
+/// several pads normally share a line.
+///
+/// Stale latched status is cleared first. A pad that saw an edge before its interrupt was armed
+/// otherwise fires immediately on enable, which looks exactly like a real event.
+pub fn setInterrupt(pin: u8, t: IntrType, line: IntrLine) void {
+ std.debug.assert(pin <= max_pin);
+ clearInterrupt(pin);
+ pin_cfg.at(pin).modify(.{
+ int_type.is(@intFromEnum(t)),
+ int_ena.is(if (t == .disable) 0 else @as(u32, 1) << @intFromEnum(line)),
+ });
+}
+
+/// Disarm, leaving the trigger type alone so it can be re-enabled unchanged.
+pub fn disableInterrupt(pin: u8) void {
+ pin_cfg.at(pin).modify(.{int_ena.is(0)});
+}
+
+pub fn interruptPending(pin: u8, line: IntrLine) bool {
+ const b = Bank.of(pin);
+ const i: u32 = @intFromEnum(line);
+ return b.pick(intr_status.at(i), intr_status1.at(i)).raw() & b.mask() != 0;
+}
+
+/// Every pad currently interrupting on `line`, as a 57-bit mask in two halves. One read of each
+/// register, so a handler can dispatch the whole set without re-reading between pads.
+pub fn pendingMask(line: IntrLine) struct { low: u32, high: u32 } {
+ const i: u32 = @intFromEnum(line);
+ return .{ .low = intr_status.at(i).raw(), .high = intr_status1.at(i).raw() };
+}
+
+pub fn clearInterrupt(pin: u8) void {
+ const b = Bank.of(pin);
+ b.pick(status_w1tc, status1_w1tc).writeRaw(b.mask());
+}
+
+pub fn clearInterrupts(low: u32, high: u32) void {
+ if (low != 0) status_w1tc.writeRaw(low);
+ if (high != 0) status1_w1tc.writeRaw(high);
+}
+
+/// Everything a pin needs to be a plain push-pull output, in the order the hardware wants: select
+/// the pad's function before enabling the driver, so the pin never spends a moment driven by
+/// whatever peripheral the IO MUX happened to be pointing at.
+pub fn configureOutput(pin: u8, opts: struct {
+ drive: Drive = .medium,
+ /// Enable the input buffer too, so the pin can be read back.
+ readback: bool = false,
+}) void {
+ setFunction(pin, .gpio);
+ // Point the matrix at the GPIO peripheral: a pad left routed to whatever signal was there
+ // before is the failure this line prevents.
+ func_out_sel.at(pin).modify(.{ out_sel.is(matrix_gpio_signal), oen_sel.is(0) });
+ pad.at(pin).modify(.{
+ fun_drv.is(@intFromEnum(opts.drive)),
+ fun_ie.is(@intFromBool(opts.readback)),
+ fun_wpu.is(0),
+ fun_wpd.is(0),
+ });
+ outputEnable(pin);
+}
+
+/// A plain input: driver off, input buffer on, optional pull.
+pub fn configureInput(pin: u8, opts: struct { pull: Pull = .none }) void {
+ outputDisable(pin);
+ setFunction(pin, .gpio);
+ pad.at(pin).modify(.{
+ fun_ie.is(1),
+ fun_wpu.is(@intFromBool(opts.pull == .up)),
+ fun_wpd.is(@intFromBool(opts.pull == .down)),
+ });
+}
+
+// ------------------------------------------------------------------------------- GPIO matrix
+
+/// The GPIO matrix: 256 peripheral output signals, any of which can be routed to any pad. This is
+/// how a UART reaches a pin that has no direct IO MUX function for it.
+const func_out_sel = mmio.RegArray(
+ regs.GPIO_FUNC0_OUT_SEL_CFG_REG,
+ regs.GPIO_FUNC1_OUT_SEL_CFG_REG,
+ pin_count,
+);
+// The input side of the matrix, indexed by *signal* rather than by pad: GPIO_FUNCn_IN_SEL_CFG
+// selects which pad feeds peripheral input signal n. That is the opposite indexing from
+// `func_out_sel` above, and it is why the two arrays exist separately.
+//
+// The base is FUNC1's address minus one word, not FUNC1's address. gpio_struct.h:849 declares
+// `func_in_sel_cfg[256]` and notes func0 is reserved, so ESP-IDF's register header defines no
+// GPIO_FUNC0_IN_SEL_CFG_REG at all - the array starts at +0x158 with a name-less word. Anchoring
+// on FUNC1 with a count of 256 is off by one in both directions: `at(n)` would configure signal
+// n+1, and `at(255)` would land on GPIO_FUNC0_OUT_SEL_CFG_REG (+0x558) and start driving a pad.
+// Bounds checked against the headers: FUNC255_IN_SEL_CFG_REG is +0x554 = 0x158 + 4*255.
+const func_in_sel = mmio.RegArray(
+ regs.GPIO_FUNC1_IN_SEL_CFG_REG - 4,
+ regs.GPIO_FUNC1_IN_SEL_CFG_REG,
+ 256,
+);
+
+const out_sel = Field.of(regs.GPIO_FUNC0_OUT_SEL_S, regs.GPIO_FUNC0_OUT_SEL_V);
+const oen_sel = Field.of(regs.GPIO_FUNC0_OEN_SEL_S, regs.GPIO_FUNC0_OEN_SEL_V);
+
+// The input side's three fields. All of GPIO_FUNCn_IN_SEL_CFG's fields share these shifts, so as
+// with the pad registers one macro triple describes all 256.
+const in_sel = Field.of(regs.GPIO_FUNC1_IN_SEL_S, regs.GPIO_FUNC1_IN_SEL_V);
+const in_inv_sel = Field.of(regs.GPIO_FUNC1_IN_INV_SEL_S, regs.GPIO_FUNC1_IN_INV_SEL_V);
+/// 1 = take this signal from the GPIO matrix, 0 = from the pad's direct IO MUX function.
+const sig_in_sel = Field.of(regs.GPIO_SIG1_IN_SEL_S, regs.GPIO_SIG1_IN_SEL_V);
+
+/// Writing this value instead of a peripheral signal index means "the GPIO peripheral drives this
+/// pad", which is the matrix's way of expressing plain GPIO output. It comes from IDF's own signal
+/// map because it is chip-specific: 256 here, 128 on the ESP32-S3.
+pub const matrix_gpio_signal: u32 = regs.SIG_GPIO_OUT_IDX;
+
+/// Route a peripheral output signal to a pad through the matrix, and let that peripheral own the
+/// pad's output enable.
+///
+/// `OEN_SEL` reads backwards from its name, and the differential test against ESP-IDF's LL is what
+/// caught it: 1 means "use GPIO_ENABLE_REG[n] as the output enable", 0 means "use the peripheral's
+/// own output enable signal" (gpio_reg.h, GPIO_FUNC0_OEN_SEL). A routed peripheral must have 0 - its
+/// OE is part of the signal being routed. The first version of this function set 1 and then set the
+/// matching GPIO_ENABLE bit to compensate, which worked by the wrong mechanism and left the pad
+/// latently output-enabled: clear OEN_SEL later and the pin would start driving on its own.
+pub fn matrixOut(pin: u8, signal: u32) void {
+ std.debug.assert(pin <= max_pin);
+ setFunction(pin, .gpio);
+ func_out_sel.at(pin).modify(.{ out_sel.is(signal), oen_sel.is(0) });
+}
+
+/// Route a pad to a peripheral *input* signal through the matrix.
+///
+/// Indexed by signal, not by pin, which is the opposite of `matrixOut`: one pad may feed any number
+/// of input signals, but a signal has exactly one source. The three writes are one word, where
+/// gpio_ll.h:613-618 uses three bitfield stores; the resulting word is identical and nothing here
+/// depends on the intermediate states, whereas a driver that read the register back between them
+/// could observe a signal sourced from the wrong pad.
+///
+/// This does not enable the pad's input buffer - `setInputEnable` does, and a routed input with
+/// `fun_ie` clear reads as a constant. Callers that want the pad readable must do both.
+pub fn matrixIn(pin: u8, signal: u32) void {
+ std.debug.assert(pin <= max_pin or pin == matrix_const_zero or pin == matrix_const_one);
+ std.debug.assert(signal < 256);
+ func_in_sel.at(signal).modify(.{
+ in_sel.is(pin),
+ in_inv_sel.is(0),
+ sig_in_sel.is(1),
+ });
+}
+
+/// Where a peripheral input signal is sourced from. The read side of `matrixIn`, for a diagnostic
+/// that has to distinguish "routed to the wrong pad" from "not routed at all" - the two look the
+/// same from the peripheral's end.
+pub const MatrixIn = struct {
+ /// A pad index, or `matrix_const_zero`/`matrix_const_one`. Meaningless when `from_matrix` is
+ /// false: the field keeps its reset value in that case, which can read like a deliberate
+ /// tie-high and is not one.
+ pin: u8,
+ inverted: bool,
+ /// `sig_in_sel`. False means the matrix is bypassed entirely and the signal comes from the
+ /// pad's direct IO MUX function - which for a peripheral that has none is undefined.
+ from_matrix: bool,
+};
+
+pub fn matrixInSource(signal: u32) MatrixIn {
+ std.debug.assert(signal < 256);
+ const w = func_in_sel.at(signal).raw();
+ return .{
+ .pin = @intCast((w >> in_sel.shift) & in_sel.unshiftedMask()),
+ .inverted = w & in_inv_sel.mask() != 0,
+ .from_matrix = w & sig_in_sel.mask() != 0,
+ };
+}
+
+/// Two values of `matrixIn`'s `pin` that are not pins: they tie the signal to a constant level
+/// inside the matrix. `gpio_reg.h:3717-3719` documents the encoding on the register itself -
+/// "s=0-56: connect GPIO[s] to this port. s=0x3F: set this port always high level. s=0x3E: set
+/// this port always low level" - and `soc/gpio_pins.h:13-14` gives them the names ESP-IDF's
+/// drivers use. They are chip-specific: 0x38/0x30 on the ESP32, 0x1E/0x1F on the C3.
+///
+/// This is how an unwired peripheral input gets a defined level. Leaving one alone is not
+/// equivalent: `in_sel` does default to 0x3F, but `sig_in_sel` defaults to 0, which bypasses the
+/// matrix entirely and takes the signal from the pad's direct IO MUX function - which for a
+/// peripheral that has none is not a constant anything. `hal/sdmmc.zig` needs both of these for
+/// slot 1's card-detect and card-interrupt inputs.
+pub const matrix_const_one: u8 = 0x3f;
+pub const matrix_const_zero: u8 = 0x3e;
+
+test "bank arithmetic splits at 32, which is where the P4's second register begins" {
+ try std.testing.expectEqual(@as(u5, 20), Bank.of(20).bit);
+ try std.testing.expect(!Bank.of(20).high);
+ try std.testing.expectEqual(@as(u5, 0), Bank.of(32).bit);
+ try std.testing.expect(Bank.of(32).high);
+ try std.testing.expectEqual(@as(u5, 24), Bank.of(56).bit);
+ try std.testing.expectEqual(@as(u32, 1) << 24, Bank.of(56).mask());
+}
diff --git a/src/hal/i2c.zig b/src/hal/i2c.zig
new file mode 100644
index 0000000..2a1d8db
--- /dev/null
+++ b/src/hal/i2c.zig
@@ -0,0 +1,1091 @@
+//! I2C0 and I2C1 in master mode, FIFO access, no interrupts and no DMA.
+//!
+//! Slave mode, LP_I2C and the RAM (non-FIFO) access path are deliberately absent.
+//!
+//! Three things about this peripheral are not visible in the register headers, and each one is a
+//! way for a port to produce a bus that half-works:
+//!
+//! **1. The timing is a dozen registers computed from one number.** SCL low, SCL high, SCL
+//! wait-high, SDA hold, SDA sample, start hold, restart setup, stop hold, stop setup and the
+//! timeout exponent all come from a single `half_cycle` derived from the source clock and the wanted
+//! SCL frequency, and several of them are written *minus one* while two deliberately are not. The
+//! arithmetic is reproduced from ESP-IDF exactly, with the line numbers, in `Timing.calculate` and
+//! `applyTiming` below - including the parts that look like bugs and are not.
+//!
+//! **2. Nothing takes effect until `CONF_UPGATE` is written.** The timing and control registers feed
+//! a synchroniser rather than the state machine directly, so a driver that configures the block and
+//! starts a transaction without `commitConfig()` runs on the *previous* configuration. It is a
+//! write-to-trigger bit that reads back 0, so nothing about the register state afterwards shows
+//! whether it was ever written - which is exactly the kind of bug a state-comparing differential
+//! test cannot see, so it is called out here instead. ESP-IDF puts the call in the driver
+//! (`esp_driver_i2c/i2c_master.c:96`, `i2c_ll_update` at `i2c_ll.h:137-141`), not in the LL
+//! functions that write the timing.
+//!
+//! **3. The command opcode numbers changed after the original ESP32, and this chip's own register
+//! header still documents the old ones.** `i2c_struct.h:1009-1021` and `i2c_reg.h:1174-1186` say
+//! "0: RSTART, 1: WRITE, 2: READ, 3: STOP, 4: END". ESP-IDF's P4 LL says RESTART=6, WRITE=1,
+//! READ=3, STOP=2 (`i2c_ll.h:55-59`), which is what every post-ESP32 target uses (esp32c3, esp32c6
+//! and esp32p4 agree; only `esp32/include/hal/i2c_ll.h:47-51` has the numbers the P4 header's prose
+//! describes). The LL is the version the shipping driver runs on silicon, so it is the one here, and
+//! `i2c_ref.c` builds its command words from IDF's own `I2C_LL_CMD_*` macros so that the
+//! differential test would catch a wrong constant here rather than agreeing with it.
+//!
+//! And one hazard for anything that snapshots this block: **reading `I2C_DATA_REG` pops the RX
+//! FIFO.** See `Data register` below.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const gpio = @import("gpio.zig");
+const clkrst = @import("clkrst.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+/// HP I2C instances. LP_I2C is a third `i2c_dev_t` in ESP-IDF (`SOC_I2C_NUM` is 3, `soc_caps.h:310`)
+/// but it lives in the LP domain with its own clock and pad rules, and is out of scope here.
+pub const port_count: u8 = 2;
+
+/// Bytes in each direction. `i2c_ll.h:29` (`I2C_LL_FIFO_LEN`); the RAM behind it is 32 bytes at
+/// +0x100 (TX) and +0x180 (RX), reachable directly only in non-FIFO mode.
+pub const fifo_len: u8 = 32;
+
+/// Command slots. **Eight on this chip**, not sixteen: `i2c_ll.h:31` says `I2C_LL_CMD_REG_NUM 8`,
+/// `i2c_struct.h:1073` declares `command[8]`, and the register header stops at `I2C_COMD7_REG`
+/// (+0x74). ESP-IDF's own `i2c_ll_master_write_cmd_reg` doc comment claims "should be less than 16"
+/// (`i2c_ll.h:433`) - that comment is stale, and `i2c_ll_master_is_cmd_done` two hundred lines later
+/// says 8 (`i2c_ll.h:1043`). Eight slots is why the driver's long transfers end a chunk with an END
+/// opcode and continue: there is no room for a command per byte.
+pub const cmd_slots: u8 = 8;
+
+// ------------------------------------------------------------------------------------ registers
+//
+// One array per register, indexed by port. The stride is checked against I2C1's own macro rather
+// than assumed: `REG_I2C_BASE(i)` is `DR_REG_I2C0_BASE + i * 0x1000` (`soc/esp32p4/include/soc/
+// soc.h:24`), which the linker script agrees with (`esp32p4.peripherals.ld:17-18`, I2C0 =
+// 0x500C4000, I2C1 = 0x500C5000).
+
+fn portArray(comptime macro0: anytype, comptime macro1: anytype) type {
+ return mmio.RegArray(macro0, macro1, port_count);
+}
+
+const scl_low_period = portArray(regs.I2C_SCL_LOW_PERIOD_REG(0), regs.I2C_SCL_LOW_PERIOD_REG(1));
+const ctr = portArray(regs.I2C_CTR_REG(0), regs.I2C_CTR_REG(1));
+const sr = portArray(regs.I2C_SR_REG(0), regs.I2C_SR_REG(1));
+const to = portArray(regs.I2C_TO_REG(0), regs.I2C_TO_REG(1));
+const fifo_st = portArray(regs.I2C_FIFO_ST_REG(0), regs.I2C_FIFO_ST_REG(1));
+const fifo_conf = portArray(regs.I2C_FIFO_CONF_REG(0), regs.I2C_FIFO_CONF_REG(1));
+const data = portArray(regs.I2C_DATA_REG(0), regs.I2C_DATA_REG(1));
+const int_raw = portArray(regs.I2C_INT_RAW_REG(0), regs.I2C_INT_RAW_REG(1));
+const int_clr = portArray(regs.I2C_INT_CLR_REG(0), regs.I2C_INT_CLR_REG(1));
+const int_ena = portArray(regs.I2C_INT_ENA_REG(0), regs.I2C_INT_ENA_REG(1));
+const sda_hold = portArray(regs.I2C_SDA_HOLD_REG(0), regs.I2C_SDA_HOLD_REG(1));
+const sda_sample = portArray(regs.I2C_SDA_SAMPLE_REG(0), regs.I2C_SDA_SAMPLE_REG(1));
+const scl_high_period = portArray(regs.I2C_SCL_HIGH_PERIOD_REG(0), regs.I2C_SCL_HIGH_PERIOD_REG(1));
+const scl_start_hold = portArray(regs.I2C_SCL_START_HOLD_REG(0), regs.I2C_SCL_START_HOLD_REG(1));
+const scl_rstart_setup = portArray(regs.I2C_SCL_RSTART_SETUP_REG(0), regs.I2C_SCL_RSTART_SETUP_REG(1));
+const scl_stop_hold = portArray(regs.I2C_SCL_STOP_HOLD_REG(0), regs.I2C_SCL_STOP_HOLD_REG(1));
+const scl_stop_setup = portArray(regs.I2C_SCL_STOP_SETUP_REG(0), regs.I2C_SCL_STOP_SETUP_REG(1));
+const filter_cfg = portArray(regs.I2C_FILTER_CFG_REG(0), regs.I2C_FILTER_CFG_REG(1));
+const comd0 = portArray(regs.I2C_COMD0_REG(0), regs.I2C_COMD0_REG(1));
+const scl_sp_conf = portArray(regs.I2C_SCL_SP_CONF_REG(0), regs.I2C_SCL_SP_CONF_REG(1));
+
+/// First address of a port's register block, for the differential harness's window.
+pub inline fn base(port: u8) u32 {
+ std.debug.assert(port < port_count);
+ return @intCast(scl_low_period.base + scl_low_period.stride * port);
+}
+
+// I2C_CTR_REG. `trans_start`, `fsm_rst` and `conf_upgate` are write-to-trigger: they read back 0,
+// so a read-modify-write of this register does not re-trigger them.
+const sda_force_out = Field.of(regs.I2C_SDA_FORCE_OUT_S, regs.I2C_SDA_FORCE_OUT_V);
+const scl_force_out = Field.of(regs.I2C_SCL_FORCE_OUT_S, regs.I2C_SCL_FORCE_OUT_V);
+const rx_full_ack_level = Field.of(regs.I2C_RX_FULL_ACK_LEVEL_S, regs.I2C_RX_FULL_ACK_LEVEL_V);
+const ms_mode = Field.of(regs.I2C_MS_MODE_S, regs.I2C_MS_MODE_V);
+const trans_start = Field.of(regs.I2C_TRANS_START_S, regs.I2C_TRANS_START_V);
+const tx_lsb_first = Field.of(regs.I2C_TX_LSB_FIRST_S, regs.I2C_TX_LSB_FIRST_V);
+const rx_lsb_first = Field.of(regs.I2C_RX_LSB_FIRST_S, regs.I2C_RX_LSB_FIRST_V);
+const arbitration_en = Field.of(regs.I2C_ARBITRATION_EN_S, regs.I2C_ARBITRATION_EN_V);
+const fsm_rst = Field.of(regs.I2C_FSM_RST_S, regs.I2C_FSM_RST_V);
+const conf_upgate = Field.of(regs.I2C_CONF_UPGATE_S, regs.I2C_CONF_UPGATE_V);
+
+// I2C_SR_REG, all read-only.
+const resp_rec = Field.of(regs.I2C_RESP_REC_S, regs.I2C_RESP_REC_V);
+const arb_lost = Field.of(regs.I2C_ARB_LOST_S, regs.I2C_ARB_LOST_V);
+const bus_busy = Field.of(regs.I2C_BUS_BUSY_S, regs.I2C_BUS_BUSY_V);
+const rxfifo_cnt = Field.of(regs.I2C_RXFIFO_CNT_S, regs.I2C_RXFIFO_CNT_V);
+const txfifo_cnt = Field.of(regs.I2C_TXFIFO_CNT_S, regs.I2C_TXFIFO_CNT_V);
+
+// I2C_TO_REG. `time_out_value` is only five bits wide - the timeout is 2^value source-clock cycles,
+// so 31 is the largest legal exponent and the arithmetic below never approaches it.
+const time_out_value = Field.of(regs.I2C_TIME_OUT_VALUE_S, regs.I2C_TIME_OUT_VALUE_V);
+const time_out_en = Field.of(regs.I2C_TIME_OUT_EN_S, regs.I2C_TIME_OUT_EN_V);
+
+// I2C_FIFO_CONF_REG. `rx_fifo_rst`/`tx_fifo_rst` are annotated R/W, not self-clearing: they hold
+// the FIFO in reset until written back to 0, which is why resetting one is two stores.
+const rxfifo_wm_thrhd = Field.of(regs.I2C_RXFIFO_WM_THRHD_S, regs.I2C_RXFIFO_WM_THRHD_V);
+const txfifo_wm_thrhd = Field.of(regs.I2C_TXFIFO_WM_THRHD_S, regs.I2C_TXFIFO_WM_THRHD_V);
+const nonfifo_en = Field.of(regs.I2C_NONFIFO_EN_S, regs.I2C_NONFIFO_EN_V);
+const rx_fifo_rst = Field.of(regs.I2C_RX_FIFO_RST_S, regs.I2C_RX_FIFO_RST_V);
+const tx_fifo_rst = Field.of(regs.I2C_TX_FIFO_RST_S, regs.I2C_TX_FIFO_RST_V);
+const fifo_prt_en = Field.of(regs.I2C_FIFO_PRT_EN_S, regs.I2C_FIFO_PRT_EN_V);
+
+// Timing fields. Every period is nine bits ([8:0], max 511) except `scl_wait_high_period`, which is
+// seven ([15:9], max 127) and shares its register with `scl_high_period`.
+const scl_low_period_f = Field.of(regs.I2C_SCL_LOW_PERIOD_S, regs.I2C_SCL_LOW_PERIOD_V);
+const scl_high_period_f = Field.of(regs.I2C_SCL_HIGH_PERIOD_S, regs.I2C_SCL_HIGH_PERIOD_V);
+const scl_wait_high_period_f = Field.of(regs.I2C_SCL_WAIT_HIGH_PERIOD_S, regs.I2C_SCL_WAIT_HIGH_PERIOD_V);
+const sda_hold_time = Field.of(regs.I2C_SDA_HOLD_TIME_S, regs.I2C_SDA_HOLD_TIME_V);
+const sda_sample_time = Field.of(regs.I2C_SDA_SAMPLE_TIME_S, regs.I2C_SDA_SAMPLE_TIME_V);
+const scl_start_hold_time = Field.of(regs.I2C_SCL_START_HOLD_TIME_S, regs.I2C_SCL_START_HOLD_TIME_V);
+const scl_rstart_setup_time = Field.of(regs.I2C_SCL_RSTART_SETUP_TIME_S, regs.I2C_SCL_RSTART_SETUP_TIME_V);
+const scl_stop_hold_time = Field.of(regs.I2C_SCL_STOP_HOLD_TIME_S, regs.I2C_SCL_STOP_HOLD_TIME_V);
+const scl_stop_setup_time = Field.of(regs.I2C_SCL_STOP_SETUP_TIME_S, regs.I2C_SCL_STOP_SETUP_TIME_V);
+
+// I2C_FILTER_CFG_REG. Both thresholds are four bits, both filters default *enabled* with a
+// threshold of 0 - which filters nothing - so "disable" and "enable with 0" are different words.
+const scl_filter_thres = Field.of(regs.I2C_SCL_FILTER_THRES_S, regs.I2C_SCL_FILTER_THRES_V);
+const sda_filter_thres = Field.of(regs.I2C_SDA_FILTER_THRES_S, regs.I2C_SDA_FILTER_THRES_V);
+const scl_filter_en = Field.of(regs.I2C_SCL_FILTER_EN_S, regs.I2C_SCL_FILTER_EN_V);
+const sda_filter_en = Field.of(regs.I2C_SDA_FILTER_EN_S, regs.I2C_SDA_FILTER_EN_V);
+
+// I2C_SCL_SP_CONF_REG: the hardware bus-clear generator.
+const scl_rst_slv_en = Field.of(regs.I2C_SCL_RST_SLV_EN_S, regs.I2C_SCL_RST_SLV_EN_V);
+const scl_rst_slv_num = Field.of(regs.I2C_SCL_RST_SLV_NUM_S, regs.I2C_SCL_RST_SLV_NUM_V);
+
+/// Data register offset in words, for the harness's `no_read` list. See `Data register` below.
+pub const data_word_offset: u32 = (0x1c - 0x00) / 4;
+
+// ----------------------------------------------------------------------------- clocks and reset
+//
+// I2C has clock control in two places, and the split is not symmetrical between the two ports:
+//
+// * the APB bus clock gate and the block reset are in HP_SYS_CLKRST's shared registers, and live
+// in `clkrst.zig` with every other peripheral's (`i2c_ll.h:149-176`);
+// * the *controller* clock - the one the bus state machine runs on - its source select and its
+// divider are I2C-specific fields of HP_SYS_CLKRST_PERI_CLK_CTRL10/11, and are here.
+//
+// The asymmetry is the trap: I2C1's source select and controller-clock enable are in PERI_CLK_CTRL10
+// beside I2C0's (bits 26 and 27, `i2c_ll.h:851-852` and `i2c_ll.h:944-945`), while I2C1's *divider*
+// is in PERI_CLK_CTRL11 (`i2c_ll.h:196-199`). Reading the field names alone would put all of I2C1
+// in ctrl11.
+
+const peri_clk_ctrl10 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL10_REG);
+const peri_clk_ctrl11 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL11_REG);
+
+const i2c0_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_V);
+const i2c1_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_V);
+const i2c0_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_V);
+const i2c1_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_V);
+const i2c0_div_num = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_V);
+const i2c0_div_numerator = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_V);
+const i2c0_div_denominator = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_V);
+const i2c1_div_num = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_V);
+const i2c1_div_numerator = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_V);
+const i2c1_div_denominator = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_V);
+
+/// Controller clock source. Two choices on this chip (`clk_tree_defs.h:486-494`), and the register
+/// field is one bit: 0 = XTAL, 1 = RC_FAST (`i2c_ll.h:848-852`).
+pub const Source = enum(u1) {
+ /// 40 MHz on this board, and the default. Accurate, which for a bus with a specified maximum
+ /// clock is the whole point.
+ xtal = 0,
+ /// The internal RC oscillator, ~20 MHz and temperature-dependent. Usable only because I2C is a
+ /// clocked bus with no baud-rate agreement to keep.
+ rc_fast = 1,
+};
+
+/// XTAL frequency on this board, as the source frequency to hand `Timing.calculate` for
+/// `Source.xtal`. Fixed by the crystal, not by the clock tree: 40 MHz.
+pub const xtal_hz: u32 = 40_000_000;
+
+/// Select the controller clock source. A read-modify-write of a register shared with the other
+/// port's clock fields, so it takes the interrupt guard.
+pub fn setSource(port: u8, src: Source) void {
+ std.debug.assert(port < port_count);
+ const v: u32 = @intFromEnum(src);
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ peri_clk_ctrl10.modify(.{if (port == 0) i2c0_clk_src_sel.is(v) else i2c1_clk_src_sel.is(v)});
+}
+
+/// The controller clock gate, which is *not* the APB gate in `clkrst.zig`: registers stay readable
+/// and writable with this off, and only the bus state machine stops. It defaults to 0
+/// (`hp_sys_clkrst_reg.h`, REG_I2C0_CLK_EN default 0), so unlike most peripherals on this chip I2C
+/// genuinely needs this call before it will do anything. `_i2c_hal_init` (`i2c_hal.c:52-58`) is
+/// where ESP-IDF makes it.
+pub fn setControllerClockEnabled(port: u8, on: bool) void {
+ std.debug.assert(port < port_count);
+ const v: u32 = @intFromBool(on);
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ peri_clk_ctrl10.modify(.{if (port == 0) i2c0_clk_en.is(v) else i2c1_clk_en.is(v)});
+}
+
+// -------------------------------------------------------------------------------------- timing
+
+/// Everything the bus timing registers need, in source-clock cycles, as ESP-IDF computes it.
+///
+/// The field widths are ESP-IDF's: `i2c_hal_clk_config_t` is nine `uint16_t`
+/// (`hal/i2c_types.h:46-56`). That matters at the extremes - a value that would exceed 65535 wraps
+/// there too - and it is why this is `u16` rather than `u32`.
+pub const Timing = struct {
+ /// Controller clock divider, as a *count*: the register takes this minus one.
+ clkm_div: u16,
+ scl_low: u16,
+ scl_high: u16,
+ scl_wait_high: u16,
+ sda_hold: u16,
+ sda_sample: u16,
+ /// Both the start-condition and the stop-condition setup time.
+ setup: u16,
+ /// Both the start-condition and the stop-condition hold time.
+ hold: u16,
+ /// Timeout *exponent*: the bus times out after 2^tout source-clock cycles.
+ tout: u16,
+
+ /// Reproduce `i2c_ll_master_cal_bus_clk` (`i2c_ll.h:104-128`) exactly.
+ ///
+ /// The whole derivation, because every line of it is load-bearing:
+ ///
+ /// clkm_div = source / (bus * 1024) + 1
+ /// sclk = source / clkm_div
+ /// half = sclk / bus / 2
+ ///
+ /// The `+ 1` is not rounding, it is a floor: the period registers are nine bits, so `half` must
+ /// stay under 512, and dividing the source clock until `sclk <= 1024 * bus` is what guarantees
+ /// it. At 40 MHz that makes `clkm_div` 1 for every bus frequency above 39 kHz and grows it
+ /// below - 10 kHz gives `clkm_div` 4, `sclk` 10 MHz, `half` 500 - so the divider is not an
+ /// optional refinement, it is what makes slow buses representable at all.
+ ///
+ /// From `half`, in source-clock cycles:
+ ///
+ /// scl_low = half
+ /// scl_wait_high = half/2 - 2 if bus >= 80 kHz, else half/4
+ /// scl_high = half - scl_wait_high
+ /// sda_hold = half/4
+ /// sda_sample = half/2
+ /// setup = hold = half
+ /// tout = 32 - clz(5 * half) + 2
+ ///
+ /// `scl_wait_high` is the part of the high period during which the master waits for the slave to
+ /// release SCL (clock stretching); `scl_high` is the part it drives. They sum to `half`, so the
+ /// nominal frequency is the same either way, and IDF's own comment (`i2c_ll.h:112-114`) records
+ /// why the split changes at 80 kHz: below that, too much wait-high measurably *raises* the
+ /// frequency on real hardware.
+ ///
+ /// The `tout` expression is `log2(5 * half) + 2` written with a count-leading-zeros: a timeout
+ /// of about 20 half-cycles, i.e. ten bit times, rounded up to the next power of two because the
+ /// register holds an exponent. IDF writes it as
+ /// `sizeof(half_cycle) * 8 - __builtin_clz(5 * half_cycle) + 2` with `half_cycle` a `uint32_t`,
+ /// hence the 32 here.
+ ///
+ /// Not reproduced: the `HAL_ASSERT` at `i2c_ll.h:126-127` that
+ /// `scl_wait_high < sda_sample < scl_high`. It holds for every frequency this can be asked for
+ /// at 40 MHz (checked from 10 kHz to 1 MHz), and an assert that cannot fire is noise; the
+ /// ordering it protects is a hardware requirement, not something this code can choose.
+ pub fn calculate(source_hz: u32, bus_hz: u32) Timing {
+ std.debug.assert(bus_hz > 0);
+ std.debug.assert(source_hz / 2 > bus_hz);
+
+ const clkm_div: u32 = source_hz / (bus_hz * 1024) + 1;
+ const sclk_hz: u32 = source_hz / clkm_div;
+ const half: u32 = sclk_hz / bus_hz / 2;
+
+ const wait_high: u32 = if (bus_hz >= 80_000) half / 2 - 2 else half / 4;
+ return .{
+ .clkm_div = @truncate(clkm_div),
+ .scl_low = @truncate(half),
+ .scl_wait_high = @truncate(wait_high),
+ .scl_high = @truncate(half - wait_high),
+ .sda_hold = @truncate(half / 4),
+ .sda_sample = @truncate(half / 2),
+ .setup = @truncate(half),
+ .hold = @truncate(half),
+ // @clz(0) is 32 in Zig where __builtin_clz(0) is undefined in C, so this differs from
+ // IDF only for half == 0, which the assert above rules out.
+ .tout = @truncate(32 - @clz(5 * half) + 2),
+ };
+ }
+};
+
+/// Write a computed `Timing` to the peripheral's ten timing registers and the controller-clock
+/// divider - `i2c_ll_master_set_bus_timing` (`i2c_ll.h:190-220`).
+///
+/// **Which values are written minus one and which are not is the substance of this function.**
+/// Eight of the ten are `value - 1`, because the hardware counts from zero. `scl_high_period` and
+/// `scl_wait_high_period` are written as-is, and that asymmetry is deliberate: IDF's comment
+/// (`i2c_ll.h:201-205`) says the Technical Reference Manual asks for minus one on those two as well,
+/// and that following it measurably produces an SCL a little *faster* than asked for, so they do not
+/// subtract. A port that "fixes" this by making all ten consistent gets a bus that is out of spec at
+/// the top end and passes every test that does not include an oscilloscope.
+///
+/// Subtractions are done in `u32` with wrapping and truncated by the field write, which is what the
+/// C does for a `uint16_t` of 0 as well - it is unreachable here anyway, since `calculate` asserts
+/// `half >= 1`.
+pub fn applyTiming(port: u8, t: Timing) void {
+ std.debug.assert(port < port_count);
+ setClockDivider(port, t.clkm_div);
+
+ scl_low_period.at(port).modify(.{scl_low_period_f.is(@as(u32, t.scl_low) -% 1)});
+ // One store where IDF does two read-modify-writes of the same register (`i2c_ll.h:207-208`).
+ // Same final word; a write-trace comparison sees the difference, a state comparison does not.
+ scl_high_period.at(port).modify(.{
+ scl_high_period_f.is(t.scl_high),
+ scl_wait_high_period_f.is(t.scl_wait_high),
+ });
+ sda_hold.at(port).modify(.{sda_hold_time.is(@as(u32, t.sda_hold) -% 1)});
+ sda_sample.at(port).modify(.{sda_sample_time.is(@as(u32, t.sda_sample) -% 1)});
+ scl_rstart_setup.at(port).modify(.{scl_rstart_setup_time.is(@as(u32, t.setup) -% 1)});
+ scl_stop_setup.at(port).modify(.{scl_stop_setup_time.is(@as(u32, t.setup) -% 1)});
+ scl_start_hold.at(port).modify(.{scl_start_hold_time.is(@as(u32, t.hold) -% 1)});
+ scl_stop_hold.at(port).modify(.{scl_stop_hold_time.is(@as(u32, t.hold) -% 1)});
+ to.at(port).modify(.{ time_out_value.is(t.tout), time_out_en.is(1) });
+}
+
+/// Compute and apply the timing for a target SCL frequency. The whole point of the file.
+///
+/// Does **not** commit: call `commitConfig` when the rest of the configuration is in place. That is
+/// ESP-IDF's division too - `_i2c_hal_set_bus_timing` (`i2c_hal.c:27-32`) is calculate-then-write,
+/// and the driver commits separately.
+pub fn setBusTiming(port: u8, source_hz: u32, bus_hz: u32) void {
+ applyTiming(port, Timing.calculate(source_hz, bus_hz));
+}
+
+/// The controller clock divider: register field is the divider *minus one*, with the fractional
+/// numerator and denominator zeroed because ESP-IDF does not use them
+/// (`i2c_ll.h:193-199`, `i2c_ll.h:229-239`).
+pub fn setClockDivider(port: u8, clkm_div: u16) void {
+ std.debug.assert(port < port_count);
+ const num: u32 = @as(u32, clkm_div) -% 1;
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ if (port == 0) {
+ peri_clk_ctrl10.modify(.{
+ i2c0_div_num.is(num),
+ i2c0_div_numerator.is(0),
+ i2c0_div_denominator.is(0),
+ });
+ } else {
+ peri_clk_ctrl11.modify(.{
+ i2c1_div_num.is(num),
+ i2c1_div_numerator.is(0),
+ i2c1_div_denominator.is(0),
+ });
+ }
+}
+
+// The three narrow timing setters, for tuning one condition without recomputing the whole set - a
+// slow slave that needs a longer SDA hold, say.
+//
+// **These do not use the same convention as `applyTiming`, and that is ESP-IDF's inconsistency, not
+// a transcription error.** `i2c_ll_master_set_start_timing` writes `scl_rstart_setup = setup` but
+// `scl_start_hold = hold - 1` (`i2c_ll.h:452-456`); `i2c_ll_master_set_stop_timing` writes both as
+// given (`i2c_ll.h:467-471`); `i2c_ll_set_sda_timing` writes both as given (`i2c_ll.h:482-486`).
+// `i2c_ll_master_set_bus_timing`, meanwhile, subtracts one from all six of those
+// (`i2c_ll.h:210-217`). The reconciliation is that `cal_bus_clk` produces *cycle counts* and these
+// setters take *register values*, with the single exception of `start_hold` - and IDF's own getters
+// agree: `i2c_ll_get_start_timing` adds one back to the hold and not to the setup
+// (`i2c_ll.h:644-648`), while `i2c_ll_get_stop_timing` adds nothing (`i2c_ll.h:659-663`). Anything
+// tidier here would be a different peripheral configuration from the one IDF produces.
+
+pub fn setStartTiming(port: u8, setup: u32, hold: u32) void {
+ std.debug.assert(port < port_count);
+ scl_rstart_setup.at(port).modify(.{scl_rstart_setup_time.is(setup)});
+ scl_start_hold.at(port).modify(.{scl_start_hold_time.is(hold -% 1)});
+}
+
+pub fn setStopTiming(port: u8, setup: u32, hold: u32) void {
+ std.debug.assert(port < port_count);
+ scl_stop_setup.at(port).modify(.{scl_stop_setup_time.is(setup)});
+ scl_stop_hold.at(port).modify(.{scl_stop_hold_time.is(hold)});
+}
+
+pub fn setSdaTiming(port: u8, sample: u32, hold: u32) void {
+ std.debug.assert(port < port_count);
+ sda_hold.at(port).modify(.{sda_hold_time.is(hold)});
+ sda_sample.at(port).modify(.{sda_sample_time.is(sample)});
+}
+
+/// Timeout exponent for a wanted timeout in microseconds -
+/// `i2c_ll_calculate_timeout_us_to_reg_val` (`i2c_ll.h:1060-1065`).
+///
+/// `32 - clz(cycles_per_us * timeout_us)` is `log2` rounded *up*, which is the only sensible
+/// direction for a bus timeout. IDF's own default for the SCL timeout is 2000 us
+/// (`i2c_ll.h:88`).
+pub fn timeoutExponent(source_hz: u32, timeout_us: u32) u32 {
+ const cycles_per_us = source_hz / 1_000_000;
+ return 32 - @clz(cycles_per_us * timeout_us);
+}
+
+/// Set just the timeout exponent, leaving the enable bit alone - `i2c_ll_set_tout`
+/// (`i2c_ll.h:358-361`). The field is five bits: 2^31 source cycles is the longest expressible
+/// timeout, which at 40 MHz is 54 seconds.
+pub fn setTimeout(port: u8, exponent: u32) void {
+ std.debug.assert(port < port_count);
+ to.at(port).modify(.{time_out_value.is(exponent)});
+}
+
+pub fn setTimeoutEnabled(port: u8, on: bool) void {
+ std.debug.assert(port < port_count);
+ to.at(port).modify(.{time_out_en.is(@intFromBool(on))});
+}
+
+/// Glitch filter: pulses shorter than `cycles` source-clock cycles are ignored on both SDA and SCL.
+/// `cycles == 0` disables both filters - `i2c_ll_master_set_filter` (`i2c_ll.h:753-764`).
+///
+/// Note what "disable" means here: the two enable bits default to 1 with thresholds of 0, so the
+/// reset state is "filtering enabled, filtering nothing", and disabling is not the same word as
+/// enabling with a threshold of 0. Passing 0 therefore leaves the thresholds untouched, exactly as
+/// IDF does, rather than zeroing them - a difference the register comparison would catch.
+pub fn setFilter(port: u8, cycles: u4) void {
+ std.debug.assert(port < port_count);
+ const r = filter_cfg.at(port);
+ if (cycles > 0) {
+ r.modify(.{
+ scl_filter_thres.is(cycles),
+ sda_filter_thres.is(cycles),
+ scl_filter_en.is(1),
+ sda_filter_en.is(1),
+ });
+ } else {
+ r.modify(.{ scl_filter_en.is(0), sda_filter_en.is(0) });
+ }
+}
+
+// ----------------------------------------------------------------------------------- bring-up
+
+/// Put a port into master mode with the defaults ESP-IDF's `i2c_hal_master_init` establishes
+/// (`i2c_hal.c:39-50`), in the same order.
+///
+/// The four control bits are one store where IDF does five separate read-modify-writes of the same
+/// register; the resulting word is identical. Each one matters:
+///
+/// * `ms_mode = 1` - master.
+/// * `sda_force_out = scl_force_out = 0` - open drain. The names are inverted:
+/// `i2c_ll_enable_pins_open_drain` writes `!enable_od` (`i2c_ll.h:971-975`), so *zero* is
+/// open-drain and one is push-pull. Push-pull on a shared bus is a short circuit the moment two
+/// devices disagree, so this is the bit that must not be got backwards.
+/// * `arbitration_en = 0` - IDF's master init disables arbitration, which defaults to 1. With a
+/// single master there is nothing to arbitrate, and a false arbitration-lost abort on a noisy
+/// line is worse than none.
+/// * `rx_full_ack_level = 0` - ACK, not NACK, when the RX FIFO hits its threshold.
+/// * `tx_lsb_first = rx_lsb_first = 0` - MSB first, which is what I2C is.
+///
+/// Then both FIFOs are reset, as IDF does, so the block starts with empty FIFOs whatever the
+/// previous user left behind.
+pub fn initMaster(port: u8) void {
+ std.debug.assert(port < port_count);
+ ctr.at(port).modify(.{
+ ms_mode.is(1),
+ sda_force_out.is(0),
+ scl_force_out.is(0),
+ arbitration_en.is(0),
+ rx_full_ack_level.is(0),
+ tx_lsb_first.is(0),
+ rx_lsb_first.is(0),
+ });
+ resetTxFifo(port);
+ resetRxFifo(port);
+}
+
+/// Latch the configuration into the state machine. Write-to-trigger, self-clearing, and required:
+/// see note 2 in this file's header. `i2c_ll_update` (`i2c_ll.h:137-141`).
+pub inline fn commitConfig(port: u8) void {
+ ctr.at(port).modify(.{conf_upgate.is(1)});
+}
+
+/// Reset the master state machine without touching its configuration. Self-clearing in hardware -
+/// IDF writes 1 and never writes 0 (`i2c_ll.h:785-789`, "fsm_rst is a self cleared bit"). For a
+/// master that has hung mid-transaction; the bus itself may still need `clearBus`.
+pub inline fn resetFsm(port: u8) void {
+ ctr.at(port).modify(.{fsm_rst.is(1)});
+}
+
+/// Drive up to `pulses` SCL clocks to free a slave that is holding SDA low, then a STOP -
+/// `i2c_ll_master_clr_bus` (`i2c_ll.h:803-810`). Nine pulses is IDF's default
+/// (`I2C_LL_RESET_SLV_SCL_PULSE_NUM_DEFAULT`, `i2c_ll.h:87`): enough for any slave to finish the
+/// byte it is stuck in and see a NACK.
+///
+/// The enable bit is cleared *by hardware* when the pulses have been sent, so completion is polled
+/// through `isBusClearDone`, and `commitConfig` is needed both to start it and, per IDF's comment,
+/// to resynchronise afterwards. Only meaningful with SCL and SDA actually routed to pads.
+pub fn clearBus(port: u8, pulses: u5) void {
+ std.debug.assert(port < port_count);
+ scl_sp_conf.at(port).modify(.{ scl_rst_slv_num.is(pulses), scl_rst_slv_en.is(1) });
+ commitConfig(port);
+}
+
+pub inline fn isBusClearDone(port: u8) bool {
+ return scl_sp_conf.at(port).get(scl_rst_slv_en) == 0;
+}
+
+/// Open-drain or push-pull SCL and SDA, at the peripheral end.
+///
+/// **The register fields are the inverse of this argument.** `i2c_ll_enable_pins_open_drain` writes
+/// `sda_force_out = scl_force_out = !enable_od` (`i2c_ll.h:971-975`), so a zero in either field is
+/// what makes that line release instead of driving high. `initMaster` already establishes
+/// open-drain; this exists to be able to change it, and to have the polarity checked against IDF's
+/// on its own rather than only as part of a seven-field store.
+///
+/// This is the *peripheral's* driver behaviour. The pad also has an open-drain bit of its own in the
+/// GPIO block (`gpio.setOpenDrain`), and a real bus needs both: the pad hardware must not drive
+/// high, and the peripheral must not ask it to.
+pub fn setPinsOpenDrain(port: u8, open_drain: bool) void {
+ std.debug.assert(port < port_count);
+ const v: u32 = @intFromBool(!open_drain);
+ ctr.at(port).modify(.{ sda_force_out.is(v), scl_force_out.is(v) });
+}
+
+// --------------------------------------------------------------------------------------- FIFOs
+
+/// FIFO or RAM access. FIFO mode is `nonfifo_en = 0`, i.e. the field is the inverse of the name of
+/// this function - `i2c_ll_enable_fifo_mode` (`i2c_ll.h:345-348`).
+pub fn setFifoMode(port: u8, fifo: bool) void {
+ std.debug.assert(port < port_count);
+ fifo_conf.at(port).modify(.{nonfifo_en.is(@intFromBool(!fifo))});
+}
+
+/// Hold the TX FIFO in reset, then release it. Two stores, because the bit is plain R/W and not
+/// self-clearing: writing only the 1 leaves the FIFO permanently reset and every subsequent
+/// transmission silently empty (`i2c_ll.h:248-253`).
+pub fn resetTxFifo(port: u8) void {
+ std.debug.assert(port < port_count);
+ const r = fifo_conf.at(port);
+ r.modify(.{tx_fifo_rst.is(1)});
+ r.modify(.{tx_fifo_rst.is(0)});
+}
+
+pub fn resetRxFifo(port: u8) void {
+ std.debug.assert(port < port_count);
+ const r = fifo_conf.at(port);
+ r.modify(.{rx_fifo_rst.is(1)});
+ r.modify(.{rx_fifo_rst.is(0)});
+}
+
+/// FIFO watermark thresholds, and the two side effects ESP-IDF attaches to setting them.
+///
+/// `fifo_prt_en` gates the watermark interrupts *and* the overflow/underflow protection
+/// (`i2c_reg.h:449-459`), and IDF sets it in both threshold setters
+/// (`i2c_ll.h:496-500` and `i2c_ll.h:510-515`), so it is set here rather than left to the caller.
+///
+/// The other side effect is less obvious and is copied deliberately: IDF's
+/// `i2c_ll_set_rxfifo_full_thr` also writes `ctr.rx_full_ack_level = 0`, in a different register.
+/// That is coherent rather than sloppy - an RX threshold means "ACK up to here", and a master that
+/// NACKed at the threshold would end the transfer instead of pausing it - but it means this
+/// operation touches two registers, and after a peripheral reset (where `rx_full_ack_level` defaults
+/// to 1) leaving it out is an observable difference rather than a stylistic one.
+pub fn setFifoThresholds(port: u8, tx_empty: u5, rx_full: u5) void {
+ std.debug.assert(port < port_count);
+ fifo_conf.at(port).modify(.{
+ fifo_prt_en.is(1),
+ txfifo_wm_thrhd.is(tx_empty),
+ rxfifo_wm_thrhd.is(rx_full),
+ });
+ ctr.at(port).modify(.{rx_full_ack_level.is(0)});
+}
+
+// ------------------------------------------------------------------------------- Data register
+//
+// **Reading `I2C_DATA_REG` pops the RX FIFO.** The register header does not say so - it annotates
+// the single field `I2C_FIFO_RDATA` as `HRO` and describes the register as "Rx FIFO read data"
+// (`i2c_reg.h:464-474`) - but ESP-IDF's LL settles it: `i2c_ll_read_rxfifo` reads *the same address*
+// `len` times into successive bytes of a buffer (`i2c_ll.h:691-697`), which can only produce
+// distinct bytes if each read advances the FIFO. The write direction is the same address for the
+// other FIFO: `i2c_ll_write_txfifo` stores `len` bytes to `hw->data.val` (`i2c_ll.h:674-680`). One
+// address, two FIFOs, both with side effects - the same shape as `UART_FIFO_REG`, and the reason
+// this offset is in the differential harness's `no_read` list.
+
+/// Push bytes into the TX FIFO. In FIFO mode each store is one byte into the FIFO regardless of the
+/// width of the access; the FIFO is `fifo_len` deep and there is no flow control here, so the caller
+/// must not exceed `txSpace`.
+pub fn writeTxFifo(port: u8, bytes: []const u8) void {
+ std.debug.assert(port < port_count);
+ std.debug.assert(bytes.len <= fifo_len);
+ const r = data.at(port);
+ for (bytes) |b| r.writeRaw(b);
+}
+
+/// Pop bytes out of the RX FIFO. Destructive by construction - see above.
+pub fn readRxFifo(port: u8, out: []u8) void {
+ std.debug.assert(port < port_count);
+ const r = data.at(port);
+ for (out) |*b| b.* = @truncate(r.raw());
+}
+
+/// Bytes waiting in the RX FIFO.
+pub inline fn rxCount(port: u8) u32 {
+ return sr.at(port).get(rxfifo_cnt);
+}
+
+/// Bytes queued in the TX FIFO.
+pub inline fn txCount(port: u8) u32 {
+ return sr.at(port).get(txfifo_cnt);
+}
+
+/// Room left in the TX FIFO, saturating at 0 the way `i2c_ll_get_txfifo_len` does
+/// (`i2c_ll.h:604-608`) - the counter can read `fifo_len` and the subtraction must not wrap.
+pub inline fn txSpace(port: u8) u32 {
+ const used = txCount(port);
+ return if (used >= fifo_len) 0 else fifo_len - used;
+}
+
+pub inline fn isBusBusy(port: u8) bool {
+ return sr.at(port).get(bus_busy) == 1;
+}
+
+// -------------------------------------------------------------------------------- command list
+//
+// A transaction is up to eight commands written into I2C_COMD0..7 and then triggered as a unit. The
+// register header exposes each slot as a single 14-bit field `I2C_COMMANDn` plus a `_DONE` bit at 31
+// and stops there: the sub-fields exist only in `i2c_ll_hw_cmd_t` (`i2c_ll.h:41-52`). So this is one
+// of the few places where the field geometry cannot come from a macro pair, and the comptime check
+// below is what keeps that honest - the five sub-fields must tile exactly the bits the header calls
+// I2C_COMMANDn.
+
+const cmd_byte_num = Field.of(0, 0xff);
+const cmd_ack_en = Field.bit(8);
+const cmd_ack_exp = Field.bit(9);
+const cmd_ack_val = Field.bit(10);
+const cmd_op_code = Field.of(11, 0x7);
+const cmd_done = Field.of(regs.I2C_COMMAND0_DONE_S, regs.I2C_COMMAND0_DONE_V);
+
+comptime {
+ const command_field = Field.of(regs.I2C_COMMAND0_S, regs.I2C_COMMAND0_V);
+ const tiled = cmd_byte_num.mask() | cmd_ack_en.mask() | cmd_ack_exp.mask() |
+ cmd_ack_val.mask() | cmd_op_code.mask();
+ if (tiled != command_field.mask()) @compileError(
+ "the command sub-fields from i2c_ll.h do not tile I2C_COMMAND0 - one of the two headers moved",
+ );
+ if (cmd_done.mask() & command_field.mask() != 0) @compileError("command done bit overlaps the command");
+}
+
+/// Opcodes, from `i2c_ll.h:55-59`. **Not** the numbers this chip's own register header describes -
+/// see note 3 in the file header.
+pub const Op = enum(u3) {
+ write = 1,
+ stop = 2,
+ read = 3,
+ /// Hand the command list back to software with the bus still held, so the next chunk can be
+ /// loaded. This is how a transfer longer than eight commands or 32 bytes is done without DMA.
+ end = 4,
+ /// START, and equally a repeated START.
+ restart = 6,
+};
+
+/// One command slot as a value rather than a raw word.
+///
+/// The three ACK fields only mean something for one direction each, which is why they are separate
+/// rather than one "ack" number:
+///
+/// * `ack_check` (WRITE) - compare the ACK bit the slave returns against `ack_expected` and abort
+/// the list if it differs. This is what turns a missing device into a NACK error instead of a
+/// transfer into the void.
+/// * `ack_value` (READ) - the ACK bit this master sends after each byte it reads. Zero (ACK) for
+/// every byte but the last, one (NACK) for the last, which is how a slave is told to stop
+/// driving the bus.
+pub const Command = struct {
+ op: Op,
+ /// Bytes to move. Only WRITE and READ use it; a READ of n bytes is one command, not n.
+ bytes: u8 = 0,
+ ack_check: bool = false,
+ ack_expected: u1 = 0,
+ ack_value: u1 = 0,
+
+ pub inline fn encode(self: Command) u32 {
+ return (@as(u32, self.bytes) << cmd_byte_num.shift) |
+ (@as(u32, @intFromBool(self.ack_check)) << cmd_ack_en.shift) |
+ (@as(u32, self.ack_expected) << cmd_ack_exp.shift) |
+ (@as(u32, self.ack_value) << cmd_ack_val.shift) |
+ (@as(u32, @intFromEnum(self.op)) << cmd_op_code.shift);
+ }
+};
+
+/// One command slot. The slot stride is checked against the header's own COMD1 macro rather than
+/// assumed to be 4.
+inline fn cmdReg(port: u8, slot: u8) Reg {
+ std.debug.assert(slot < cmd_slots);
+ const stride = comptime mmio.addr(regs.I2C_COMD1_REG(0)) - mmio.addr(regs.I2C_COMD0_REG(0));
+ comptime {
+ // ... and the array is contiguous all the way to the last slot.
+ if (mmio.addr(regs.I2C_COMD7_REG(0)) != mmio.addr(regs.I2C_COMD0_REG(0)) + stride * 7)
+ @compileError("the command registers are not a contiguous array of 8");
+ }
+ return Reg.atAddress(comd0.at(port).address + stride * slot);
+}
+
+/// Write a command into a slot. A whole-word store, as IDF's `i2c_ll_master_write_cmd_reg` does
+/// (`i2c_ll.h:437-441`): it is the one register here where establishing the entire word is right,
+/// because the `done` bit must go back to 0 for the slot to be waited on again.
+pub fn writeCommand(port: u8, slot: u8, cmd: Command) void {
+ std.debug.assert(port < port_count);
+ cmdReg(port, slot).writeRaw(cmd.encode());
+}
+
+/// Load a whole command list, in order. Any slot the list does not reach keeps whatever it held -
+/// which is harmless, because the sequencer stops at the STOP or END that the list must contain.
+pub fn writeCommands(port: u8, cmds: []const Command) void {
+ std.debug.assert(cmds.len <= cmd_slots);
+ for (cmds, 0..) |c, i| writeCommand(port, @intCast(i), c);
+}
+
+/// Whether the sequencer has finished a slot. Set by hardware (`R/W/SS`), cleared by writing the
+/// slot again. `i2c_ll_master_is_cmd_done` (`i2c_ll.h:1047-1051`).
+pub inline fn isCommandDone(port: u8, slot: u8) bool {
+ return cmdReg(port, slot).get(cmd_done) == 1;
+}
+
+// --------------------------------------------------------------------------------- transactions
+
+// The master event bits, in I2C_INT_RAW/I2C_INT_ST/I2C_INT_CLR - the same bit numbers in all three
+// (`i2c_ll.h:61-70`). Reading INT_RAW is safe: the bits are `R/SS/WTC`, set by hardware and cleared
+// only by writing a 1 to the same position in INT_CLR, so polling does not consume them. Writing
+// INT_CLR is the one place in this file that must be `writeRaw` rather than `modify`.
+const int_trans_complete = Field.of(regs.I2C_TRANS_COMPLETE_INT_RAW_S, regs.I2C_TRANS_COMPLETE_INT_RAW_V);
+const int_end_detect = Field.of(regs.I2C_END_DETECT_INT_RAW_S, regs.I2C_END_DETECT_INT_RAW_V);
+const int_nack = Field.of(regs.I2C_NACK_INT_RAW_S, regs.I2C_NACK_INT_RAW_V);
+const int_arbitration_lost = Field.of(regs.I2C_ARBITRATION_LOST_INT_RAW_S, regs.I2C_ARBITRATION_LOST_INT_RAW_V);
+const int_time_out = Field.of(regs.I2C_TIME_OUT_INT_RAW_S, regs.I2C_TIME_OUT_INT_RAW_V);
+const int_scl_st_to = Field.of(regs.I2C_SCL_ST_TO_INT_RAW_S, regs.I2C_SCL_ST_TO_INT_RAW_V);
+const int_scl_main_st_to = Field.of(regs.I2C_SCL_MAIN_ST_TO_INT_RAW_S, regs.I2C_SCL_MAIN_ST_TO_INT_RAW_V);
+
+/// The mask ESP-IDF uses for "all interrupts" - `I2C_LL_INTR_MASK`, `i2c_ll.h:1097`.
+///
+/// It is 14 bits, and this block has 19 (`I2C_SLAVE_ADDR_UNMATCH_INT` is bit 18). The five it leaves
+/// out are slave-mode and general-call events, which is presumably why IDF's mask stops where it
+/// does; the value is IDF's rather than a recount so that clearing "everything" means the same thing
+/// on both sides of the differential.
+pub const all_interrupts: u32 = 0x3fff;
+
+/// Clear interrupt flags. Write-1-to-clear, so this is a raw store of a mask and never a
+/// read-modify-write: reading INT_RAW and writing it back would clear whatever had arrived in
+/// between and nothing else.
+pub inline fn clearInterrupts(port: u8, mask: u32) void {
+ int_clr.at(port).writeRaw(mask);
+}
+
+/// Mask every interrupt at the peripheral. This HAL polls; nothing here reaches the CLIC.
+///
+/// A whole-word zero rather than IDF's `int_ena &= ~mask` (`i2c_ll.h:305-309`), so it also covers
+/// the five slave-mode bits outside `all_interrupts`. Reaching the same word from a block whose
+/// `int_ena` reset value is 0 either way, which is why the differential case for it agrees.
+pub inline fn disableInterrupts(port: u8) void {
+ int_ena.at(port).writeRaw(0);
+}
+
+/// How a triggered command list ended.
+pub const Outcome = enum {
+ /// The list ran to its STOP.
+ complete,
+ /// The list hit an END opcode: the bus is still held and the next chunk can be loaded.
+ end_detect,
+ /// A slave did not acknowledge. The usual meaning is "nothing at that address".
+ nack,
+ /// Another master won the bus. Only possible with `arbitration_en` set, which `initMaster`
+ /// clears.
+ arbitration_lost,
+ /// SCL was held low past the configured timeout - `I2C_TO_REG`. Almost always a slave holding
+ /// the clock, or no pull-up on the line at all.
+ timeout,
+ /// The SCL state machine stalled: `scl_st_to` or `scl_main_st_to`. IDF's driver treats this as
+ /// the signal that a bus deadlock may have happened and `clearBus` is worth trying
+ /// (`i2c_ll.h:795`).
+ stalled,
+ /// Nothing had happened yet.
+ pending,
+};
+
+/// Trigger the loaded command list. Write-to-trigger; the bit reads back 0, so this leaves no trace
+/// in a register snapshot. `i2c_ll_start_trans` (`i2c_ll.h:629-633`).
+pub inline fn startTransaction(port: u8) void {
+ ctr.at(port).modify(.{trans_start.is(1)});
+}
+
+/// Read the outcome so far from one load of INT_RAW.
+///
+/// Errors are reported ahead of completion, and in the order they matter: an arbitration loss or a
+/// NACK can be raised in the same word as `trans_complete`, and calling that transaction complete
+/// is how a driver comes to believe a device answered when it did not.
+pub fn outcome(port: u8) Outcome {
+ const raw = int_raw.at(port).raw();
+ if (raw & int_arbitration_lost.mask() != 0) return .arbitration_lost;
+ if (raw & int_nack.mask() != 0) return .nack;
+ if (raw & int_time_out.mask() != 0) return .timeout;
+ if (raw & (int_scl_st_to.mask() | int_scl_main_st_to.mask()) != 0) return .stalled;
+ if (raw & int_trans_complete.mask() != 0) return .complete;
+ if (raw & int_end_detect.mask() != 0) return .end_detect;
+ return .pending;
+}
+
+/// Spin until the transaction resolves. Returns `.pending` if it never does, rather than hanging:
+/// a bus with no pull-up produces exactly that, and it is a fault to report rather than a board to
+/// power-cycle.
+///
+/// `spins` is a loop count, not a time. At the ~90 MHz this board boots at, a 100 kHz transfer of a
+/// few bytes needs on the order of 10^4 iterations of this loop; the default of 200,000 leaves an
+/// order of magnitude of headroom and still returns in well under a second.
+pub fn waitTransaction(port: u8, spins: u32) Outcome {
+ var n: u32 = 0;
+ while (n < spins) : (n += 1) {
+ const o = outcome(port);
+ if (o != .pending) return o;
+ }
+ return .pending;
+}
+
+/// The status register's own error bits, which are not the interrupt flags: `resp_rec` is the last
+/// ACK level *received* and `arb_lost` is the state machine's own latch. Both are read-only and
+/// survive an interrupt clear, so they are what to look at when diagnosing a transfer after the fact.
+pub const Status = struct {
+ /// The ACK bit the slave last returned: 0 = ACK, 1 = NACK.
+ last_ack: u1,
+ arbitration_lost: bool,
+ bus_busy: bool,
+ rx_bytes: u32,
+ tx_bytes: u32,
+};
+
+pub fn status(port: u8) Status {
+ const raw = sr.at(port).raw();
+ return .{
+ .last_ack = @intCast((raw >> resp_rec.shift) & 1),
+ .arbitration_lost = raw & arb_lost.mask() != 0,
+ .bus_busy = raw & bus_busy.mask() != 0,
+ .rx_bytes = (raw >> rxfifo_cnt.shift) & rxfifo_cnt.unshiftedMask(),
+ .tx_bytes = (raw >> txfifo_cnt.shift) & txfifo_cnt.unshiftedMask(),
+ };
+}
+
+// ------------------------------------------------------------------------------------ the pads
+//
+// I2C is a two-wire open-drain bus and the P4 reaches it only through the GPIO matrix: there is no
+// IO MUX function for I2C on any pad, so both signals go out through `matrixOut` and come back in
+// through `matrixIn`. Both directions are needed even for a write-only master - the master samples
+// SDA to read the slave's ACK, and samples SCL to detect stretching - which is why every pad here
+// gets its input buffer enabled as well as its driver.
+
+/// The GPIO matrix signal indices for a port, from ESP-IDF's own signal map
+/// (`gpio_sig_map.h:141-148`) via `i2c_periph.c`. On this chip a signal's input and output index
+/// happen to be the same number, which is not true on every part and is not something to rely on.
+pub fn sclSignal(port: u8) u32 {
+ return switch (port) {
+ 0 => regs.I2C0_SCL_PAD_OUT_IDX,
+ else => regs.I2C1_SCL_PAD_OUT_IDX,
+ };
+}
+
+pub fn sdaSignal(port: u8) u32 {
+ return switch (port) {
+ 0 => regs.I2C0_SDA_PAD_OUT_IDX,
+ else => regs.I2C1_SDA_PAD_OUT_IDX,
+ };
+}
+
+/// Route SCL and SDA to two pads, open-drain, following `i2c_common_set_pins`
+/// (`esp_driver_i2c/i2c_common.c:318-345`) step for step.
+///
+/// **The internal pull-ups are not enough for a real bus.** They are on the order of 45 kOhm, which
+/// with a few tens of picofarads of trace and device capacitance gives a rise time far past the
+/// 1 us that 100 kHz I2C allows. ESP-IDF says the same thing in its own driver documentation and
+/// enables them anyway as a convenience for a single device on a short wire. A bus that is expected
+/// to work needs external resistors - 4.7 kOhm to 3.3 V is the usual choice at 100 kHz, 2.2 kOhm at
+/// 400 kHz - and then `internal_pullups` should be false, because two resistors in parallel is not
+/// what either calculation assumed.
+///
+/// The order matters in one place: the pad is driven high *before* its output is enabled, so
+/// enabling the driver cannot pull the bus low for the few cycles before the peripheral takes over.
+/// A low SCL glitch is a clock edge to every device on the bus.
+pub fn configurePins(port: u8, scl_pin: u8, sda_pin: u8, opts: struct {
+ internal_pullups: bool = false,
+}) void {
+ std.debug.assert(port < port_count);
+ for ([_]struct { pin: u8, signal: u32 }{
+ .{ .pin = scl_pin, .signal = sclSignal(port) },
+ .{ .pin = sda_pin, .signal = sdaSignal(port) },
+ }) |wire| {
+ gpio.setHigh(wire.pin);
+ gpio.setInputEnable(wire.pin, true);
+ gpio.setOpenDrain(wire.pin, true);
+ gpio.setPull(wire.pin, if (opts.internal_pullups) .up else .none);
+ gpio.matrixOut(wire.pin, wire.signal);
+ gpio.matrixIn(wire.pin, wire.signal);
+ }
+}
+
+// ------------------------------------------------------------------------------- transfers
+
+/// Bring a port up as a master on a given bus frequency, in the order the hardware requires:
+/// clocks, then reset, then configuration, then commit.
+///
+/// Reset before configure, because a reset drops everything configured before it. `clkrst.init`
+/// does the gate-then-reset pair; the controller clock is separate and enabled after, since it only
+/// feeds the state machine.
+pub fn init(port: u8, opts: struct {
+ source: Source = .xtal,
+ source_hz: u32 = xtal_hz,
+ bus_hz: u32 = 100_000,
+ /// Glitch filter width in source-clock cycles. ESP-IDF's driver default is 7.
+ filter_cycles: u4 = 7,
+}) void {
+ std.debug.assert(port < port_count);
+ switch (port) {
+ 0 => clkrst.init(.i2c0),
+ else => clkrst.init(.i2c1),
+ }
+ setControllerClockEnabled(port, true);
+ setSource(port, opts.source);
+
+ initMaster(port);
+ setFifoMode(port, true);
+ disableInterrupts(port);
+ clearInterrupts(port, all_interrupts);
+ setBusTiming(port, opts.source_hz, opts.bus_hz);
+ setFilter(port, opts.filter_cycles);
+ commitConfig(port);
+}
+
+/// Default spin budget for `write`/`read`. See `waitTransaction`.
+pub const default_spins: u32 = 200_000;
+
+/// Write `bytes` to a 7-bit address as one command list.
+///
+/// RSTART | WRITE (1 + len bytes, ack checked) | STOP
+///
+/// The address byte goes in the TX FIFO ahead of the data and is counted in the WRITE command's byte
+/// count: to the sequencer the address is just the first byte written after a START. `ack_check` is
+/// on, so a missing device comes back as `.nack` rather than as a successful write into nothing.
+///
+/// One command list, one FIFO load: at most `fifo_len - 1` = 31 data bytes. Longer transfers need
+/// the END-and-continue loop that ESP-IDF's driver runs from its interrupt handler, which is out of
+/// scope here - hence the assert rather than a partial write.
+pub fn write(port: u8, address: u7, bytes: []const u8, spins: u32) Outcome {
+ std.debug.assert(bytes.len < fifo_len);
+ resetTxFifo(port);
+ resetRxFifo(port);
+ clearInterrupts(port, all_interrupts);
+
+ writeTxFifo(port, &[_]u8{@as(u8, address) << 1});
+ writeTxFifo(port, bytes);
+
+ writeCommands(port, &.{
+ .{ .op = .restart },
+ .{ .op = .write, .bytes = @intCast(bytes.len + 1), .ack_check = true },
+ .{ .op = .stop },
+ });
+ commitConfig(port);
+ startTransaction(port);
+ return waitTransaction(port, spins);
+}
+
+/// Read into `out` from a 7-bit address as one command list.
+///
+/// RSTART | WRITE 1 (address|read, ack checked) | READ n-1 sending ACK | READ 1 sending NACK | STOP
+///
+/// The last byte is a separate command because its ACK bit differs: a master that ACKs the final
+/// byte tells the slave to keep going, and the slave then holds SDA for a byte that will never be
+/// clocked out. That is the classic I2C read bug, and it is a *command list* bug - which is why the
+/// split is here rather than being something the caller can get wrong.
+///
+/// Reads of one byte collapse to a single NACKed READ, so the list is four commands instead of five.
+pub fn read(port: u8, address: u7, out: []u8, spins: u32) Outcome {
+ std.debug.assert(out.len > 0);
+ std.debug.assert(out.len <= fifo_len);
+ resetTxFifo(port);
+ resetRxFifo(port);
+ clearInterrupts(port, all_interrupts);
+
+ writeTxFifo(port, &[_]u8{(@as(u8, address) << 1) | 1});
+
+ writeCommand(port, 0, .{ .op = .restart });
+ writeCommand(port, 1, .{ .op = .write, .bytes = 1, .ack_check = true });
+ var slot: u8 = 2;
+ if (out.len > 1) {
+ writeCommand(port, slot, .{ .op = .read, .bytes = @intCast(out.len - 1), .ack_value = 0 });
+ slot += 1;
+ }
+ writeCommand(port, slot, .{ .op = .read, .bytes = 1, .ack_value = 1 });
+ writeCommand(port, slot + 1, .{ .op = .stop });
+
+ commitConfig(port);
+ startTransaction(port);
+ const result = waitTransaction(port, spins);
+ if (result == .complete) readRxFifo(port, out);
+ return result;
+}
+
+test "the timing arithmetic reproduces ESP-IDF's, including where it looks wrong" {
+ // 100 kHz on a 40 MHz XTAL: the case every I2C device supports, worked through by hand from
+ // i2c_ll.h:104-128. clkm_div = 40e6/(100e3*1024) + 1 = 0 + 1 = 1, so sclk stays 40 MHz and
+ // half = 40e6/100e3/2 = 200.
+ const t100 = Timing.calculate(40_000_000, 100_000);
+ try std.testing.expectEqual(@as(u16, 1), t100.clkm_div);
+ try std.testing.expectEqual(@as(u16, 200), t100.scl_low);
+ try std.testing.expectEqual(@as(u16, 98), t100.scl_wait_high); // half/2 - 2
+ try std.testing.expectEqual(@as(u16, 102), t100.scl_high); // half - wait_high
+ try std.testing.expectEqual(@as(u16, 50), t100.sda_hold);
+ try std.testing.expectEqual(@as(u16, 100), t100.sda_sample);
+ try std.testing.expectEqual(@as(u16, 200), t100.setup);
+ try std.testing.expectEqual(@as(u16, 200), t100.hold);
+ // 5*200 = 1000, which needs 10 bits, so 32 - 22 + 2 = 12: a timeout of 2^12 = 4096 cycles,
+ // 102 us at 40 MHz, about ten bit times.
+ try std.testing.expectEqual(@as(u16, 12), t100.tout);
+
+ // 400 kHz: same divider, quarter the half-cycle.
+ const t400 = Timing.calculate(40_000_000, 400_000);
+ try std.testing.expectEqual(@as(u16, 1), t400.clkm_div);
+ try std.testing.expectEqual(@as(u16, 50), t400.scl_low);
+ try std.testing.expectEqual(@as(u16, 23), t400.scl_wait_high);
+ try std.testing.expectEqual(@as(u16, 27), t400.scl_high);
+ try std.testing.expectEqual(@as(u16, 10), t400.tout);
+
+ // 10 kHz: the branch that actually uses the controller-clock divider. 40e6/(10e3*1024) = 3, so
+ // clkm_div = 4, sclk = 10 MHz and half = 500 - just inside the nine-bit period fields, which is
+ // what the divider exists to guarantee.
+ const t10 = Timing.calculate(40_000_000, 10_000);
+ try std.testing.expectEqual(@as(u16, 4), t10.clkm_div);
+ try std.testing.expectEqual(@as(u16, 500), t10.scl_low);
+ // Below 80 kHz the wait-high split changes: half/4 rather than half/2 - 2.
+ try std.testing.expectEqual(@as(u16, 125), t10.scl_wait_high);
+ try std.testing.expectEqual(@as(u16, 375), t10.scl_high);
+
+ // The hardware ordering constraint IDF asserts (i2c_ll.h:126-127) across the whole range.
+ for ([_]u32{ 10_000, 50_000, 100_000, 400_000, 1_000_000 }) |hz| {
+ const t = Timing.calculate(40_000_000, hz);
+ try std.testing.expect(t.scl_wait_high < t.sda_sample);
+ try std.testing.expect(t.sda_sample < t.scl_high);
+ // Every period register is nine bits wide, and scl_low is written minus one.
+ try std.testing.expect(t.scl_low - 1 <= 511);
+ try std.testing.expect(t.scl_wait_high <= 127); // this one is seven
+ try std.testing.expect(t.tout <= 31); // and the timeout exponent is five
+ }
+}
+
+test "the timeout exponent rounds up, and where the five-bit field runs out" {
+ // 2000 us at 40 MHz is 80,000 cycles; 2^17 = 131,072 is the first power of two above it, so
+ // IDF's documented default SCL timeout comes out as 17 - which fits the five-bit field with
+ // room to spare. This test exists because the first version of this file asserted the opposite.
+ try std.testing.expectEqual(@as(u32, 17), timeoutExponent(40_000_000, 2000));
+ try std.testing.expect(timeoutExponent(40_000_000, 2000) <= time_out_value.max());
+ // The field runs out at 2^31 source cycles, 53.7 seconds at 40 MHz - a timeout no I2C bus has a
+ // use for, which is why neither IDF nor this file range-checks it. Past that the exponent is
+ // truncated by the field write rather than rejected, exactly as IDF's bitfield store does.
+ try std.testing.expectEqual(@as(u32, 32), timeoutExponent(40_000_000, 100_000_000));
+ try std.testing.expect(timeoutExponent(40_000_000, 100_000_000) > time_out_value.max());
+}
+
+test "commands encode to the layout i2c_ll_hw_cmd_t describes" {
+ // A WRITE of three bytes with ACK checking: byte_num=3, ack_en=1, op_code=1.
+ try std.testing.expectEqual(
+ @as(u32, 3) | (1 << 8) | (1 << 11),
+ (Command{ .op = .write, .bytes = 3, .ack_check = true }).encode(),
+ );
+ // RESTART is opcode 6 on this chip, not 0 - the number the register header's prose still gives.
+ try std.testing.expectEqual(@as(u32, 6 << 11), (Command{ .op = .restart }).encode());
+ // A final READ NACKs: ack_val=1 at bit 10, opcode 3.
+ try std.testing.expectEqual(
+ @as(u32, 1) | (1 << 10) | (3 << 11),
+ (Command{ .op = .read, .bytes = 1, .ack_value = 1 }).encode(),
+ );
+ // STOP is 2 and READ is 3, which is the pair the ESP32-era numbering had the other way around.
+ try std.testing.expectEqual(@as(u32, 2 << 11), (Command{ .op = .stop }).encode());
+}
diff --git a/src/hal/intr.zig b/src/hal/intr.zig
new file mode 100644
index 0000000..38f8789
--- /dev/null
+++ b/src/hal/intr.zig
@@ -0,0 +1,965 @@
+//! The interrupt controller. The ESP32-P4 has a **CLIC**, not a PLIC and not the Xtensa-style
+//! fixed matrix of the older parts: `soc_caps.h:191` defines SOC_INT_CLIC_SUPPORTED 1, and
+//! `soc/interrupt_reg.h:16` says so in prose. Three consequences shape this file.
+//!
+//! **1. Two independent stages.** A peripheral source does not have a CPU interrupt number; it has
+//! a *mapping register*. The interrupt matrix at DR_REG_INTERRUPT_CORE0_BASE holds one 6-bit word
+//! per source, and writing `line + 16` into it points that source at external CLIC line `line`.
+//! The `+ 16` is not decoration: the CLIC's first 16 IDs are the RISC-V internal interrupts
+//! (software, timer, external), so the 32 lines a driver may use are IDs 16..47.
+//! `hal/interrupt_clic_ll.h:35-48` is the matrix write; the `+ RV_EXTERNAL_INT_OFFSET` that turns a
+//! line number into a CLIC ID is one level up, at `riscv/interrupt_clic.c:26`. Per-line control -
+//! enable, trigger, priority, pending - is the *other* stage, in the CLIC's own register file at
+//! DR_REG_CLIC_CTRL_BASE, and it is indexed by CLIC ID, i.e. by `line + 16` again.
+//!
+//! **2. The threshold is a memory-mapped register on this die, not the `mintthresh` CSR.** This is
+//! the single easiest thing to get wrong here, because every RISC-V CLIC document and every
+//! ESP32-P4 rev-3 build says `mintthresh` (CSR 0x347). `soc/interrupt_reg.h:28-40` selects
+//! `INTTHRESH_STANDARD 0` under CONFIG_ESP32P4_SELECTS_REV_LESS_V3 - the same condition that
+//! selects the `register/hw_ver1` headers this project builds against - and
+//! `riscv/csr_clic.h:37-47` then leaves MINTTHRESH_CSR *undefined*. The threshold lives in
+//! CLIC_INT_THRESH_REG at 0x2080_0008, bits [31:24] (`soc/clic_reg.h:61-67`). Writing CSR 0x347 on
+//! this silicon is not an illegal instruction and not an error; it writes a register the interrupt
+//! arbiter does not read, so interrupts stay masked and nothing says why.
+//!
+//! **3. `regs.INTTHRESH_STANDARD` lies, and must not be used.** The register module is
+//! `zig translate-c` over the headers with *no* sdkconfig, so CONFIG_ESP32P4_SELECTS_REV_LESS_V3 is
+//! absent there and `interrupt_reg.h` takes its `#else` branch: the translated module contains
+//! `pub const INTTHRESH_STANDARD = 1`, which is the wrong answer for this die. (The oracle's C side
+//! is compiled against `src/oracle/oracle_sdkconfig.h:25`, which does define it, so IDF's own code
+//! there takes the correct branch. The two disagree, deliberately, and only the C side is right
+//! about this macro.) Nothing in this file reads it.
+//!
+//! Nothing below has been run on hardware by the author of this file. What is claimed is that the
+//! register arithmetic matches ESP-IDF's at the cited lines, and that `src/oracle/intr_cases.zig`
+//! compares the two on the die. Taking an actual interrupt is a behavioural property no register
+//! comparison can establish; see the note at the foot of that file.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const clkrst = @import("clkrst.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+// ------------------------------------------------------------------------------- geometry
+
+/// CLIC IDs 0..15 are the RISC-V internal interrupts; a driver cannot have them. IDs 16..47 are the
+/// 32 external lines. `riscv/csr_clic.h:28-29` (RV_EXTERNAL_INT_COUNT, RV_EXTERNAL_INT_OFFSET) and
+/// `soc/clic_reg.h:14` (CLIC_EXT_INTR_NUM_OFFSET) are three names for these two numbers.
+pub const line_count: u32 = 32;
+pub const ext_offset: u32 = @intCast(regs.CLIC_EXT_INTR_NUM_OFFSET);
+/// 16 internal + 32 external. `hal/interrupt_clic_ll.h:22` RV_TOTAL_INT_COUNT, and the hardware
+/// agrees: CLIC_INT_INFO_REG's NUM_INT field reads 48 at reset (`soc/clic_reg.h:54-59`).
+pub const total_ids: u32 = 48;
+
+/// Priority levels. `soc/clic_reg.h:13` NLBITS 3, so 8 levels, held in the *top* 3 bits of the
+/// 8-bit CLIC_INT_CTL field. Level 0 is masked by the reset threshold; a usable interrupt wants 1
+/// or more.
+pub const NLBITS: u5 = @intCast(regs.NLBITS);
+const nlbits_shift: u5 = 8 - NLBITS;
+/// The low `8 - NLBITS` bits of a priority/threshold byte are not part of the level and IDF fills
+/// them with ones (`riscv/csr_clic.h:59`, NLBITS_TO_BYTE). Reproduced exactly, because the
+/// differential compares the whole word.
+const nlbits_pad: u32 = (@as(u32, 1) << nlbits_shift) - 1;
+
+// -------------------------------------------------------------------------- interrupt matrix
+
+/// Every peripheral interrupt source on this chip, from `soc/interrupts.h` - which opens with
+/// "This table is decided by hardware, don't touch this."
+///
+/// IDs 0..127 are contiguous and each has a mapping register at `matrix_base + 4*id`: the last of
+/// them, `assist_debug` = 127, is INTERRUPT_CORE0_ASSIST_DEBUG_INT_MAP_REG at +0x1FC, which is
+/// exactly 4*127. That is the invariant `interrupt_clic_ll.h:46` depends on when it computes the
+/// address arithmetically rather than from a table.
+///
+/// **The last three exist only on chip revision >= 3.0 and therefore not on this die.**
+/// `soc/interrupts.h:155-160` explains the gap: their mapping registers are *not* contiguous with
+/// the rest, so IDF gave them IDs 133-135 to make `base + 4*id` land on the right address anyway.
+/// The numbering hole at 128..132 is that workaround, not missing hardware. On a pre-v3 part -
+/// which is what `regs.ZIG_P4_HW_VER == 1` asserts - routing one of them writes a register that
+/// nothing drives.
+pub const Source = enum(u8) {
+ lp_rtc = 0,
+ lp_wdt = 1,
+ lp_timer_reg0 = 2,
+ lp_timer_reg1 = 3,
+ mb_hp = 4,
+ mb_lp = 5,
+ pmu_0 = 6,
+ pmu_1 = 7,
+ lp_anaperi = 8,
+ lp_adc = 9,
+ lp_gpio = 10,
+ lp_i2c = 11,
+ lp_i2s = 12,
+ lp_spi = 13,
+ lp_touch = 14,
+ /// Also spelled ETS_TEMPERATURE_SENSOR_INTR_SOURCE; IDF aliases the two (`interrupts.h:34`).
+ lp_tsens = 15,
+ lp_uart = 16,
+ lp_efuse = 17,
+ lp_sw = 18,
+ lp_sysreg = 19,
+ lp_huk = 20,
+ sys_icm = 21,
+ usb_serial_jtag = 22,
+ sdio_host = 23,
+ dw_gdma = 24,
+ spi2 = 25,
+ spi3 = 26,
+ i2s0 = 27,
+ i2s1 = 28,
+ i2s2 = 29,
+ uhci0 = 30,
+ uart0 = 31,
+ uart1 = 32,
+ uart2 = 33,
+ uart3 = 34,
+ uart4 = 35,
+ lcd_cam = 36,
+ adc = 37,
+ pwm0 = 38,
+ pwm1 = 39,
+ twai0 = 40,
+ twai1 = 41,
+ twai2 = 42,
+ rmt = 43,
+ i2c0 = 44,
+ i2c1 = 45,
+ tg0_t0 = 46,
+ tg0_t1 = 47,
+ tg0_wdt_level = 48,
+ tg1_t0 = 49,
+ tg1_t1 = 50,
+ tg1_wdt_level = 51,
+ ledc = 52,
+ systimer_target0 = 53,
+ systimer_target1 = 54,
+ systimer_target2 = 55,
+ ahb_pdma_in_ch0 = 56,
+ ahb_pdma_in_ch1 = 57,
+ ahb_pdma_in_ch2 = 58,
+ ahb_pdma_out_ch0 = 59,
+ ahb_pdma_out_ch1 = 60,
+ ahb_pdma_out_ch2 = 61,
+ axi_pdma_in_ch0 = 62,
+ axi_pdma_in_ch1 = 63,
+ axi_pdma_in_ch2 = 64,
+ axi_pdma_out_ch0 = 65,
+ axi_pdma_out_ch1 = 66,
+ axi_pdma_out_ch2 = 67,
+ rsa = 68,
+ aes = 69,
+ sha = 70,
+ ecc = 71,
+ ecdsa = 72,
+ km = 73,
+ gpio_intr0 = 74,
+ gpio_intr1 = 75,
+ gpio_intr2 = 76,
+ gpio_intr3 = 77,
+ gpio_pad_comp = 78,
+ from_cpu_intr0 = 79,
+ from_cpu_intr1 = 80,
+ from_cpu_intr2 = 81,
+ from_cpu_intr3 = 82,
+ cache = 83,
+ mspi = 84,
+ csi_bridge = 85,
+ dsi_bridge = 86,
+ csi = 87,
+ dsi = 88,
+ gmii_phy = 89,
+ lpi = 90,
+ pmt = 91,
+ eth_mac = 92,
+ usb_otg = 93,
+ usb_otg_endp_multi_proc = 94,
+ jpeg = 95,
+ ppa = 96,
+ core0_trace = 97,
+ core1_trace = 98,
+ hp_core_ctrl = 99,
+ isp = 100,
+ i3c_mst = 101,
+ i3c_slv = 102,
+ usb_otg11_ch0 = 103,
+ dma2d_in_ch0 = 104,
+ dma2d_in_ch1 = 105,
+ dma2d_out_ch0 = 106,
+ dma2d_out_ch1 = 107,
+ dma2d_out_ch2 = 108,
+ psram_mspi = 109,
+ hp_sysreg = 110,
+ pcnt = 111,
+ hp_pau = 112,
+ hp_parlio_rx = 113,
+ hp_parlio_tx = 114,
+ h264_dma2d_out_ch0 = 115,
+ h264_dma2d_out_ch1 = 116,
+ h264_dma2d_out_ch2 = 117,
+ h264_dma2d_out_ch3 = 118,
+ h264_dma2d_out_ch4 = 119,
+ h264_dma2d_in_ch0 = 120,
+ h264_dma2d_in_ch1 = 121,
+ h264_dma2d_in_ch2 = 122,
+ h264_dma2d_in_ch3 = 123,
+ h264_dma2d_in_ch4 = 124,
+ h264_dma2d_in_ch5 = 125,
+ h264_reg = 126,
+ assist_debug = 127,
+
+ /// Chip rev >= 3.0 only - absent on this die. See the note above.
+ dma2d_in_ch2 = 133,
+ /// Chip rev >= 3.0 only - absent on this die.
+ dma2d_out_ch3 = 134,
+ /// Chip rev >= 3.0 only - absent on this die.
+ axi_perf_mon = 135,
+
+ /// True on a source that this pre-v3 silicon does not have.
+ pub inline fn isRev3Only(self: Source) bool {
+ return @intFromEnum(self) >= 133;
+ }
+};
+
+/// The last source ID with a mapping register on pre-v3 silicon.
+pub const max_source_id: u8 = @intFromEnum(Source.assist_debug);
+
+/// Core 0's interrupt matrix. Core 1's is 0x800 above it (`reg_base.h:198-199`) and is not reachable
+/// from here: this image runs core 0 only - core 1 is held in reset at power-on
+/// (HP_SYS_CLKRST REG_RST_EN_CORE1_GLOBAL defaults to 1) - and routing a source to a core that is
+/// not running is a way to lose an interrupt silently rather than loudly.
+const matrix_base: u32 = mmio.addr(regs.DR_REG_INTERRUPT_CORE0_BASE);
+
+/// The mapping register's only field: 6 bits, holding a CLIC ID. Taken from UART0's macro pair
+/// because the field is identical in all 128 of them - `interrupt_core0_reg.h` repeats
+/// `_INT_MAP` / mask 0x3F / shift 0 for every source. (`INTERRUPT_CORE0_*_INT_MAP_M` is one of the
+/// 153 `_M` macros that are broken C inside ESP-IDF and appear here as poisoned decls; the `_S`/`_V`
+/// pair is the only usable form, which is what `mmio.Field.of` takes.)
+const int_map = Field.of(regs.INTERRUPT_CORE0_UART0_INT_MAP_S, regs.INTERRUPT_CORE0_UART0_INT_MAP_V);
+
+inline fn mapReg(source_id: u8) Reg {
+ return Reg.atAddress(matrix_base + 4 * @as(u32, source_id));
+}
+
+/// Point a peripheral source at an external CLIC line.
+///
+/// This is only the matrix half. A routed source still needs `setEnabled(line, true)`, a trigger
+/// type, a priority above the threshold, a handler, and mstatus.MIE - `configureLine` does the
+/// CLIC-side four in the order the hardware wants.
+///
+/// Several sources may share one line; that is the normal way to fit 128 sources into 32 lines, and
+/// the handler then has to ask each peripheral whether it was the one. Nothing here prevents it.
+pub fn route(source: Source, line: u5) void {
+ routeId(@intFromEnum(source), line);
+}
+
+/// `route` by raw source ID, for a source this enum does not name.
+///
+/// The write is a read-modify-write of the low 6 bits, exactly as `interrupt_clic_ll.h:46` does it
+/// (`REG_SET_BITS(DR_REG_INTERRUPT_CORE0_BASE + 4*intr_src, intr_num, RV_INT_MASK)` with
+/// RV_INT_MASK 63 at line 25). The upper 26 bits are reserved and preserved.
+pub fn routeId(source_id: u8, line: u5) void {
+ std.debug.assert(source_id <= max_source_id);
+ mapReg(source_id).modify(.{int_map.is(@as(u32, line) + ext_offset)});
+}
+
+/// Detach a source from every line.
+///
+/// Writes CLIC ID 0, which is `ETS_INVALID_INUM` on this chip (`soc/esp32p4/include/soc/soc.h:251`)
+/// and is what `esp_system/port/cpu_start.c:185` writes into all 128 mapping registers at boot.
+/// ID 0 is an internal RISC-V interrupt line that the matrix cannot actually drive, so it means
+/// "nowhere" rather than "line 0" - note the asymmetry with `route`, which adds 16.
+pub fn unroute(source: Source) void {
+ mapReg(@intFromEnum(source)).modify(.{int_map.is(0)});
+}
+
+/// Which external line a source is routed to, or null if it is unrouted or points at an internal ID.
+pub fn routedLine(source: Source) ?u5 {
+ const id = mapReg(@intFromEnum(source)).get(int_map);
+ if (id < ext_offset or id >= ext_offset + line_count) return null;
+ return @intCast(id - ext_offset);
+}
+
+// ------------------------------------------------------------------------- per-line control
+
+/// One 32-bit control word per CLIC ID at `DR_REG_CLIC_CTRL_BASE + 4*id` (`soc/clic_reg.h:69`).
+/// Indexed by CLIC ID, so every accessor here adds `ext_offset` to the caller's line number.
+///
+/// The same word is also described byte-wise by the `BYTE_CLIC_*` macros (clic_reg.h:113-160), and
+/// ESP-IDF uses both spellings: `interrupt_clic_ll.h` does 32-bit REG_SET_FIELD, the TEE build does
+/// 8-bit stores. They land on the same bits, and each field sits wholly inside one byte, so a
+/// 32-bit read-modify-write of one field and a byte store of that byte are indistinguishable in the
+/// resulting word. This file uses the 32-bit form throughout.
+const clic_ctrl_base: u32 = mmio.addr(regs.DR_REG_CLIC_CTRL_BASE);
+
+/// Priority, bits [31:24]. Reset value 0x1f (clic_reg.h:70).
+const int_ctl = Field.of(regs.CLIC_INT_CTL_S, regs.CLIC_INT_CTL_V);
+/// Trigger type, bits [18:17].
+const int_attr_trig = Field.of(regs.CLIC_INT_ATTR_TRIG_S, regs.CLIC_INT_ATTR_TRIG_V);
+/// Hardware vectoring: 1 means fetch the handler address from MTVT rather than trapping to mtvec.
+const int_attr_shv = Field.of(regs.CLIC_INT_ATTR_SHV_S, regs.CLIC_INT_ATTR_SHV_V);
+/// Enable, bit 8.
+const int_ie = Field.of(regs.CLIC_INT_IE_S, regs.CLIC_INT_IE_V);
+/// Pending, bit 0. Read/write, with asymmetric semantics - see `edgeAck`.
+const int_ip = Field.of(regs.CLIC_INT_IP_S, regs.CLIC_INT_IP_V);
+
+inline fn ctrl(line: u5) Reg {
+ return Reg.atAddress(clic_ctrl_base + 4 * (@as(u32, line) + ext_offset));
+}
+
+/// By raw CLIC ID rather than by external line, for the one caller that has to reach the 16
+/// internal IDs: `init`, silencing everything the ROM may have left enabled.
+inline fn ctrlRegById(clic_id: u32) Reg {
+ std.debug.assert(clic_id < total_ids);
+ return Reg.atAddress(clic_ctrl_base + 4 * clic_id);
+}
+
+/// How a source drives its line. The encoding is a two-bit field whose *low* bit selects
+/// level-versus-edge and whose high bit selects the edge, which is why `interrupt_clic_ll.h:60`
+/// masks the read with `& 1` to answer "is it edge-triggered": `0b10` is a level interrupt too.
+/// (`soc/clic_reg.h:84-88`.)
+pub const Trigger = enum(u2) {
+ level = 0,
+ rising_edge = 1,
+ /// 0b10 - low bit clear, so this is a *level* trigger despite the encoding's shape. Present
+ /// only because the field is two bits wide; no source should be configured with it.
+ level_alias = 2,
+ falling_edge = 3,
+
+ pub inline fn isEdge(self: Trigger) bool {
+ return @intFromEnum(self) & 1 != 0;
+ }
+};
+
+pub fn setEnabled(line: u5, on: bool) void {
+ ctrl(line).modify(.{int_ie.is(@intFromBool(on))});
+}
+
+pub fn isEnabled(line: u5) bool {
+ return ctrl(line).get(int_ie) == 1;
+}
+
+pub fn setTrigger(line: u5, t: Trigger) void {
+ ctrl(line).modify(.{int_attr_trig.is(@intFromEnum(t))});
+}
+
+pub fn getTrigger(line: u5) Trigger {
+ return @enumFromInt(ctrl(line).get(int_attr_trig));
+}
+
+/// Priority 0..7, stored left-aligned in the 8-bit CLIC_INT_CTL field.
+///
+/// The stored byte is `priority << (8 - NLBITS)` with the low bits **zero**, which is what
+/// `esp_tee_rv_utils.h:112` writes and what `interrupt_clic_ll.h:74` reads back with `>> (8-NLBITS)`.
+/// Note the asymmetry with the *threshold*, where IDF fills the same low bits with ones
+/// (`csr_clic.h:59`). Copying the threshold's encoding here would leave a different word behind
+/// than IDF's, for the same nominal priority.
+pub fn setPriority(line: u5, priority: u3) void {
+ ctrl(line).modify(.{int_ctl.is(@as(u32, priority) << nlbits_shift)});
+}
+
+pub fn getPriority(line: u5) u3 {
+ return @intCast(ctrl(line).get(int_ctl) >> nlbits_shift);
+}
+
+/// Hardware vectoring for one line. With SHV set, the CLIC jumps to `MTVT + 4*id` instead of to
+/// mtvec's base; `installVectorTable` fills every slot with the same trap entry, so flipping this
+/// changes the fetch path and not the code that runs. `interrupt_clic_ll.h:99-102`.
+pub fn setVectored(line: u5, on: bool) void {
+ ctrl(line).modify(.{int_attr_shv.is(@intFromBool(on))});
+}
+
+pub fn isVectored(line: u5) bool {
+ return ctrl(line).get(int_attr_shv) == 1;
+}
+
+pub fn isPending(line: u5) bool {
+ return ctrl(line).get(int_ip) == 1;
+}
+
+/// Acknowledge an edge-triggered interrupt.
+///
+/// Writing **1** to IP is what clears it for an edge source. That reads backwards, and clic_reg.h
+/// only hints at it - "This bit has different set and clear logic in the case of level interrupt
+/// and edge interrupt" (clic_reg.h:106-107) - but ESP-IDF's function that does exactly this store is
+/// named `rv_utils_intr_edge_ack` (`esp_private/interrupt_clic.h`, the `REG_SET_BIT(..., CLIC_INT_IP)`
+/// at the end of that header). For a *level* source this instead asserts the pending bit, which is
+/// how software raises one by hand; there is no acknowledge for a level source at the CLIC at all,
+/// the handler must clear the peripheral's own status register.
+pub fn edgeAck(line: u5) void {
+ ctrl(line).modify(.{int_ip.is(1)});
+}
+
+/// Raise a line from software. Same store as `edgeAck`; the two names exist because the hardware
+/// gives one write two meanings depending on `Trigger`.
+pub fn setPending(line: u5) void {
+ ctrl(line).modify(.{int_ip.is(1)});
+}
+
+/// Bitmask of the 32 external lines that are enabled, one loop over the control words. Mirrors
+/// `rv_utils_intr_get_enabled_mask` in `esp_private/interrupt_clic.h`.
+pub fn enabledMask() u32 {
+ var m: u32 = 0;
+ var i: u5 = 0;
+ while (true) : (i += 1) {
+ if (isEnabled(i)) m |= @as(u32, 1) << i;
+ if (i == line_count - 1) break;
+ }
+ return m;
+}
+
+// ----------------------------------------------------------------------------- the threshold
+
+/// CLIC_INT_THRESH_REG - 0x2080_0008 (`soc/clic_reg.h:61`), **not** the `mintthresh` CSR. See the
+/// module comment: on this pre-v3 die `csr_clic.h` does not even define MINTTHRESH_CSR, and a write
+/// to CSR 0x347 here is accepted and ignored.
+const thresh_reg = Reg.at(regs.CLIC_INT_THRESH_REG);
+const cpu_int_thresh = Field.of(regs.CLIC_CPU_INT_THRESH_S, regs.CLIC_CPU_INT_THRESH_V);
+
+/// Mask every interrupt whose priority is <= `level`.
+///
+/// The comparison is **inclusive**: threshold 0 lets priorities 1..7 through, threshold 7 masks
+/// everything. `esp_private/interrupt_clic.h:198-203` makes the same point when it computes
+/// `mask_int_level_lower_than(n)` as `set_intlevel(n - 1)`. Reset is 0, i.e. open.
+///
+/// Two details reproduced from IDF rather than invented:
+/// * the byte is `(level << 5) | 0x1f` - the low `8 - NLBITS` bits are filled with **ones**
+/// (`csr_clic.h:59`, NLBITS_TO_BYTE), which is the opposite of the per-line priority encoding;
+/// * the register is read back immediately afterwards. That is not a paranoid verification, it is
+/// ordering: `esp_private/interrupt_clic.h:139-144` records that the CPU does not see the new
+/// threshold until the store has actually left the write buffer, and that a load - or about
+/// eight nops - is what forces it. Without the load, re-enabling mstatus.MIE on the next
+/// instruction can take an interrupt the new threshold was meant to mask.
+///
+/// `write` rather than `modify` is deliberate and matches IDF's `REG_WRITE`: CLIC_CPU_INT_THRESH is
+/// the register's only field, so there is nothing to preserve.
+pub fn setThreshold(level: u3) void {
+ thresh_reg.write(.{cpu_int_thresh.is((@as(u32, level) << nlbits_shift) | nlbits_pad)});
+ _ = thresh_reg.raw();
+}
+
+pub fn getThreshold() u3 {
+ return @intCast(thresh_reg.get(cpu_int_thresh) >> nlbits_shift);
+}
+
+// ------------------------------------------------------------- vector table and trap entry
+
+/// CSR numbers, from `components/riscv/include/riscv/csr_clic.h`:
+/// * `MTVT_CSR 0x307` (line 34) - base of the interrupt jump table.
+/// * `MTVEC_MODE_CSR 3` (line 22) - the two low bits of mtvec that put the core in CLIC mode.
+/// * `MINTSTATUS_CSR 0x346` (`soc/interrupt_reg.h:36`) - **non-standard on this die**; the RISC-V
+/// CLIC specification and IDF's rev-3 path both say 0xFB1 (`csr_clic.h:40`).
+/// * `MINTTHRESH_CSR 0x347` exists only when INTTHRESH_STANDARD is 1, which it is not here.
+pub const mtvt_csr = 0x307;
+pub const mintstatus_csr = 0x346;
+pub const mtvec_mode_clic = 3;
+/// mstatus.MIE. Same bit `clkrst.Guard` manipulates.
+const mstatus_mie: u32 = 1 << 3;
+
+/// A line's handler. Runs with mstatus.MIE clear - this file does not implement nesting - on the
+/// interrupted stack, so it must be short and must not use floating point: `trapEntry` saves the
+/// integer caller-saved registers and nothing else, and `_start` leaves the FPU enabled, so a
+/// handler that touches an f-register corrupts whatever it interrupted.
+pub const Handler = *const fn (line: u5) void;
+
+var handlers: [line_count]?Handler = @splat(null);
+
+/// Interrupts that arrived on a line with no handler, or on one of the 16 internal CLIC IDs. Not
+/// reset by anything here: a non-zero value after a run is the diagnostic.
+pub var spurious: u32 = 0;
+
+/// The CLIC's jump table: one address per CLIC ID, internal and external.
+///
+/// 48 entries, and 256-byte aligned because the CLIC requires MTVT to be aligned to a power of two
+/// at least as large as the table (4 * 48 = 192 bytes, so 256). The alignment travels with the
+/// symbol, so the generated linker script's `.bss ... ALIGN(4)` is not a problem - the linker pads
+/// to the input section's own alignment. No dedicated section is needed and build.zig is unchanged.
+///
+/// Every slot points at the same `trapEntry`. A per-line stub would save the dispatch load, but it
+/// would be 48 near-identical pieces of assembly to be wrong in, and the win is a handful of cycles
+/// against a handler call. The table exists because the hardware needs one when SHV is set, not
+/// because the entries differ.
+var vector_table: [total_ids]u32 align(256) = @splat(0);
+
+/// What `init` found before it changed anything. Diagnostics, and the only record of the state the
+/// bootloader hands over in - every one of these is overwritten by `init` itself, so nothing else
+/// can observe them.
+pub var boot_state: BootState = .{};
+pub const BootState = struct {
+ /// mstatus.MIE as handed over. Measured 1 on this board, which is the fact the whole ownership
+ /// sequence below exists for.
+ mie: bool = false,
+ /// Which of the 32 external lines had CLIC_INT_IE set before `init` cleared them.
+ enabled_lines: u32 = 0,
+ /// How many of the 128 peripheral sources were pointing at an external line before `init`
+ /// detached them.
+ routed_sources: u32 = 0,
+};
+
+/// Take ownership of the interrupt controller, then point it at this file.
+///
+/// **The bootloader hands over with interrupts globally enabled.** Measured: `mie_at_boot=1`. That
+/// single fact is why this function is a sequence rather than three CSR writes, and it cost two
+/// silent hangs to establish. Two separate hazards follow from it, and clearing MIE only fixes the
+/// first:
+///
+/// 1. `init(); attach(...)` used to take an interrupt the moment the line's IE bit went up, before
+/// the caller had said it was ready. `globalDisable()` first fixes that.
+///
+/// 2. **Whatever the ROM had armed is still armed.** The ROM ran with its own mtvec and its own
+/// reasons to enable interrupts; the matrix and the CLIC's IE bits are not reset by the handover.
+/// The instant this file's caller sets MIE, any line the ROM left enabled vectors into
+/// `trapEntry` - on an ID nothing here has a handler for. That increments `spurious` and
+/// `mret`s; and if the source is level-triggered and still asserting, the next instruction traps
+/// again, forever, with the console silent. The failure looks exactly like "our own line is not
+/// being delivered", which is what it was mistaken for.
+///
+/// So this function does what ESP-IDF's `core_intr_matrix_clear` does before it trusts the
+/// controller (`esp_system/port/cpu_start.c:174-198`), and in the same order:
+/// * detach all 128 sources by writing ETS_INVALID_INUM (cpu_start.c:183-189);
+/// * clear every line's enable, which IDF gets for free from the CLIC's reset values and this
+/// image does not, because the ROM ran first;
+/// * set every external line vectored (cpu_start.c:193-196 - "Set all the CPU interrupt lines to
+/// vectored by default, as it is on other RISC-V targets").
+///
+/// The register differential could not have found any of this: MIE is a CSR, and the boot state of
+/// the matrix is identical on both sides of every comparison because both sides inherit it.
+///
+/// Leaves MIE clear. Enabling interrupts stays the caller's decision, via `globalEnable()`.
+pub fn init() void {
+ boot_state.mie = globalEnabled();
+ globalDisable();
+
+ // Record and then silence every line, before anything can be delivered anywhere.
+ var l: u5 = 0;
+ while (true) : (l += 1) {
+ if (isEnabled(l)) boot_state.enabled_lines |= @as(u32, 1) << l;
+ if (l == line_count - 1) break;
+ }
+ // All 48 IDs, internal ones included: this core's interrupts are ours now, and an internal ID
+ // left enabled is as capable of trapping into `trapEntry` as an external one.
+ var id: u32 = 0;
+ while (id < total_ids) : (id += 1) {
+ ctrlRegById(id).modify(.{int_ie.is(0)});
+ }
+
+ // Detach every source. cpu_start.c:183-189 writes ETS_INVALID_INUM (0) to all of them.
+ var src: u32 = 0;
+ while (src <= max_source_id) : (src += 1) {
+ const r = mapReg(@intCast(src));
+ const was = r.get(int_map);
+ if (was >= ext_offset and was < ext_offset + line_count) boot_state.routed_sources += 1;
+ r.modify(.{int_map.is(0)});
+ }
+
+ const entry = @intFromPtr(&trapEntry);
+ for (&vector_table) |*slot| slot.* = @intCast(entry);
+
+ asm volatile ("csrw %[csr], %[val]"
+ :
+ : [csr] "i" (mtvt_csr),
+ [val] "r" (@as(u32, @intCast(@intFromPtr(&vector_table)))),
+ );
+ // mtvec = base | 3. Mode 3 is what `rv_utils_set_mtvec` writes (`riscv/rv_utils.h:168-171` with
+ // MTVEC_MODE_CSR from `csr_clic.h:22`) and it is what makes the core interpret mcause and MTVT
+ // as CLIC rather than as the standard vectored interface.
+ //
+ // The hardware uses `mtvec[31:6] << 6` (vectors_clic.S:38-46 spells this out), so it ignores the
+ // low six bits entirely: a `trapEntry` that were not 64-byte aligned would silently vector up to
+ // 60 bytes *before* the function. `trapEntryAddress()` exists so a test can prove on the die
+ // that it is aligned rather than trusting the linker.
+ asm volatile ("csrw mtvec, %[val]"
+ :
+ : [val] "r" (@as(u32, @intCast(entry)) | mtvec_mode_clic),
+ );
+
+ // Every external line vectored, matching cpu_start.c:193-196. Also the safer default in its own
+ // right: SHV=1 is the only delivery path ESP-IDF exercises on this chip, so it is the only one
+ // the silicon has been validated against. See `configureLine`.
+ l = 0;
+ while (true) : (l += 1) {
+ setVectored(l, true);
+ if (l == line_count - 1) break;
+ }
+
+ // Threshold open, matching IDF's RVHAL_INTR_ENABLE_THRESH of 0 (`csr_clic.h:16`): every line
+ // then gates on its own IE bit and its priority, which is where a driver can reason about it.
+ setThreshold(0);
+}
+
+/// Diagnostics a behavioural test can print, because the two facts they establish - that the trap
+/// entry is 64-byte aligned and that MTVT is 256-byte aligned - are properties of the *link*, and
+/// the shipped image is stripped, so there is no way to check them from the host.
+pub fn trapEntryAddress() u32 {
+ return @intCast(@intFromPtr(&trapEntry));
+}
+
+pub fn vectorTableAddress() u32 {
+ return @intCast(@intFromPtr(&vector_table));
+}
+
+pub fn readMtvec() u32 {
+ return asm volatile ("csrr %[out], mtvec"
+ : [out] "=r" (-> u32),
+ );
+}
+
+pub fn readMtvt() u32 {
+ return asm volatile ("csrr %[out], %[csr]"
+ : [out] "=r" (-> u32),
+ : [csr] "i" (mtvt_csr),
+ );
+}
+
+/// mintstatus, CSR 0x346 on this die (`soc/interrupt_reg.h:36`). Bits [31:24] are the current
+/// interrupt level: non-zero outside a handler would mean a previous trap never returned.
+pub fn readMintstatus() u32 {
+ return asm volatile ("csrr %[out], %[csr]"
+ : [out] "=r" (-> u32),
+ : [csr] "i" (mintstatus_csr),
+ );
+}
+
+/// Install (or, with null, remove) the handler for one external line.
+///
+/// Done with interrupts masked because the store is a pointer the trap entry may be about to load;
+/// `clkrst.maskInterrupts` composes - it restores only the MIE that was there - so this is safe to
+/// call from inside an already-masked region.
+pub fn setHandler(line: u5, handler: ?Handler) void {
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ handlers[line] = handler;
+}
+
+/// Everything one line needs, in the order the hardware wants: handler before enable, so a source
+/// that is already pending cannot reach an empty slot; trigger and priority before enable, so the
+/// first interrupt is taken under the intended configuration rather than under the reset one.
+///
+/// Does not touch the matrix - `route` is the other half - and does not touch mstatus.
+pub fn configureLine(line: u5, opts: struct {
+ handler: Handler,
+ trigger: Trigger = .level,
+ /// Must exceed the threshold to ever be taken; the threshold comparison is inclusive.
+ priority: u3 = 1,
+ /// Hardware vectoring: fetch the handler address from `MTVT + 4*id` instead of trapping to
+ /// mtvec's base.
+ ///
+ /// **On by default, and the default is the interesting part.** Every slot of the table holds the
+ /// same `trapEntry`, so this changes only how the core finds that address - which makes the
+ /// choice look free, and it is not. ESP-IDF sets SHV on all 32 lines at boot
+ /// (`cpu_start.c:193-196`, "Set all the CPU interrupt lines to vectored by default, as it is on
+ /// other RISC-V targets") and puts nothing but `j _panic_handler` at mtvec's base
+ /// (`vectors_clic.S:47-52`). So on this chip the SHV=0 delivery path is one ESP-IDF never takes
+ /// and therefore one nobody has validated. Defaulting to the path the vendor exercises is worth
+ /// more than the memory fetch it costs.
+ ///
+ /// **Measured on the die: it is the other way round, and the default is now `false`.**
+ ///
+ /// With SHV=1 the interrupt was never delivered. The core vectored to a wild address and took an
+ /// instruction access fault - `mcause=0x30000001` (EXCCODE 1, MINHV clear, so the fault was not
+ /// during the table fetch), at a `mepc` that differed run to run, with `taken=0` proving the
+ /// trap entry was never reached. mtvec, MTVT and the table contents were all verified correct
+ /// beforehand: `mtvec=0x40001383` = entry|3, `mtvt=0x4ff00100`, and every slot holding
+ /// `0x40001380` = `trapEntry`.
+ ///
+ /// The difference from ESP-IDF is *where the table lives*. IDF's `_mtvt_table` is in
+ /// `.section .exception_vectors_table.text` (`vectors_clic.S:32,67`), i.e. instruction space.
+ /// This image has no IRAM: it executes from flash through the MMU, so a table that `init()` has
+ /// to write must live in L2MEM, and the hardware vector fetch does not appear to work from
+ /// there. Since flash is not writable at run time, there is nowhere else to put it, which makes
+ /// SHV=0 the correct choice for this memory layout rather than a workaround.
+ ///
+ /// With SHV=0 both halves of the behavioural test pass: one interrupt taken, dispatched to the
+ /// right handler, `last_clic_id=21`, no spurious - and the threshold experiment then shows the
+ /// memory-mapped register at 0x2080_0008 really is the one the arbiter reads.
+ ///
+ /// `true` remains available for an image that gains an IRAM section, and the vector table is
+ /// still populated so that switching is a one-word change.
+ vectored: bool = false,
+}) void {
+ setHandler(line, opts.handler);
+ setTrigger(line, opts.trigger);
+ setPriority(line, opts.priority);
+ setVectored(line, opts.vectored);
+ setEnabled(line, true);
+}
+
+/// Route a source and bring its line up in one call.
+pub fn attach(source: Source, line: u5, opts: struct {
+ handler: Handler,
+ trigger: Trigger = .level,
+ priority: u3 = 1,
+ /// See `configureLine`: vectored is the only path ESP-IDF exercises on this chip.
+ vectored: bool = false,
+}) void {
+ route(source, line);
+ configureLine(line, .{
+ .handler = opts.handler,
+ .trigger = opts.trigger,
+ .priority = opts.priority,
+ .vectored = opts.vectored,
+ });
+}
+
+// --------------------------------------------------------------------------- global enable
+
+/// mstatus.MIE on. Nothing is taken before this, whatever the CLIC is configured to do.
+pub inline fn globalEnable() void {
+ asm volatile ("csrs mstatus, %[m]"
+ :
+ : [m] "r" (mstatus_mie),
+ );
+}
+
+pub inline fn globalDisable() void {
+ asm volatile ("csrc mstatus, %[m]"
+ :
+ : [m] "r" (mstatus_mie),
+ );
+}
+
+pub inline fn globalEnabled() bool {
+ const s = asm volatile ("csrr %[out], mstatus"
+ : [out] "=r" (-> u32),
+ );
+ return s & mstatus_mie != 0;
+}
+
+/// The composable form: mask, do something, restore whatever was there.
+///
+/// const guard = intr.mask();
+/// defer guard.release();
+///
+/// This is `clkrst.maskInterrupts` under another name, re-exported rather than reimplemented so
+/// that a critical section written against either module is the same critical section. It nests
+/// correctly - `release` only sets MIE if MIE was set on entry - which is why `setHandler` can use
+/// it without caring who called it.
+pub const Guard = clkrst.Guard;
+pub inline fn mask() Guard {
+ return clkrst.maskInterrupts();
+}
+
+// ------------------------------------------------------------------------------- trap entry
+
+/// How many times `trapEntry` has dispatched an interrupt, and the last CLIC ID it saw. Diagnostics:
+/// with `taken == 0` the trap was never reached at all, which separates "the CLIC did not deliver"
+/// from "the handler did not run".
+pub var taken: u32 = 0;
+pub var last_clic_id: u32 = 0;
+
+/// An exception - not an interrupt - that reached `trapEntry`.
+pub const Fault = struct {
+ /// Full mcause. Bit 31 is clear by construction here; the low bits are the exception code
+ /// (1 instruction access, 2 illegal instruction, 5 load access, 7 store access, 11 ecall).
+ mcause: u32,
+ /// The instruction that faulted.
+ mepc: u32,
+ /// The address or instruction word involved, per exception code.
+ mtval: u32,
+};
+
+pub var faults: u32 = 0;
+pub var last_fault: Fault = .{ .mcause = 0, .mepc = 0, .mtval = 0 };
+
+/// Called with the fault already recorded, before parking. Install one to get the numbers out;
+/// `hal` cannot print, so this hook is the only way a fault becomes visible.
+///
+/// hal.intr.on_fault = struct {
+/// fn f(x: hal.intr.Fault) void {
+/// soc.rom.print("MARK FAULT mcause=0x%08x mepc=0x%08x mtval=0x%08x\r\n",
+/// .{ x.mcause, x.mepc, x.mtval });
+/// }
+/// }.f;
+pub var on_fault: ?*const fn (Fault) void = null;
+
+/// Called from `trapEntry` with the CLIC ID out of mcause. Not part of the API; `export` because
+/// the assembly calls it by name.
+export fn intrDispatch(clic_id: u32) callconv(.c) void {
+ taken +%= 1;
+ last_clic_id = clic_id;
+ if (clic_id < ext_offset or clic_id >= ext_offset + line_count) {
+ // One of the 16 internal IDs. This file routes nothing there, so it is a bug elsewhere -
+ // most likely something the ROM left armed that `init` did not manage to silence.
+ spurious +%= 1;
+ return;
+ }
+ const line: u5 = @intCast(clic_id - ext_offset);
+ if (handlers[line]) |h| h(line) else spurious +%= 1;
+}
+
+/// The exception arm of `trapEntry`. Records, reports if a hook is installed, and **parks**.
+///
+/// Parking rather than returning is the whole point. `mret` from an exception resumes at the
+/// faulting instruction, which faults again immediately: every mistake anywhere in this file used to
+/// become an unbreakable loop through the trap entry with the console silent, indistinguishable from
+/// "the interrupt was never delivered". It cost a debugging round to tell those apart. ESP-IDF makes
+/// the same choice by putting `j _panic_handler` at mtvec's base (`vectors_clic.S:47-52`).
+export fn intrFault(mcause: u32, mepc: u32, mtval: u32) callconv(.c) noreturn {
+ faults +%= 1;
+ last_fault = .{ .mcause = mcause, .mepc = mepc, .mtval = mtval };
+ globalDisable();
+ if (on_fault) |f| f(last_fault);
+ while (true) {}
+}
+
+/// The trap entry: every trap on this core arrives here, interrupt or exception.
+///
+/// Reached three ways, and they are not interchangeable:
+/// * an **interrupt with SHV = 1**, through `MTVT + 4*id`;
+/// * an **interrupt with SHV = 0**, through mtvec's base;
+/// * an **exception**, always through mtvec's base, whatever any line's SHV says.
+///
+/// 64-byte aligned, and this is a hardware requirement rather than tidiness: in CLIC mode the core
+/// computes the target as `mtvec[31:6] << 6` (`vectors_clic.S:38-46` states it outright), so the low
+/// six bits of mtvec are not part of the address. A trap entry that were not 64-byte aligned would
+/// vector up to 60 bytes *before* this function, into whatever the linker put there. Measured in the
+/// linked image: 0x4000_1140, and `trapEntryAddress()` lets a test confirm it on the die, since the
+/// shipped image is stripped and there is no symbol to check from the host.
+///
+/// **The first thing it does is decide whether this was an interrupt at all.** mcause bit 31 says
+/// so, and getting that wrong is not a small bug: an exception whose handler `mret`s resumes at the
+/// faulting instruction and faults again, immediately and forever, with the console silent. That
+/// failure is indistinguishable from "the interrupt was never delivered", and the two were in fact
+/// confused for a debugging round. So the exception arm never returns - see `intrFault`.
+///
+/// Saves the integer caller-saved set - ra, t0-t6, a0-a7, sixteen words - and nothing else. Not
+/// saved, deliberately and with consequences:
+/// * **the f registers.** `src/main.zig`'s `_start` sets mstatus.FS to enable the FPU, so a handler
+/// that does float arithmetic silently corrupts the interrupted code. Handlers must stay integer.
+/// * **mepc, mcause, mstatus.** In CLIC mode the core stacks the previous privilege, interrupt
+/// enable and interrupt level in mcause itself, and `mret` restores them from there - so nothing
+/// here may write mcause, and nothing does. They are only at risk from a *nested* trap, and MIE
+/// stays clear for the whole sequence, so nothing can nest. That is also why there is no `mnxti`
+/// loop: the CLIC's hardware nesting (SOC_INT_HW_NESTED_SUPPORTED, `soc_caps.h:193`) is unused.
+///
+/// One consequence of `mret` worth stating because it defeats an obvious defence: it restores
+/// mstatus.MIE from MPIE, which the hardware set to 1 on entry. A handler that calls
+/// `globalDisable()` therefore does **not** leave interrupts off after it returns. To stop a runaway
+/// source the handler must clear it at the peripheral, or call `setEnabled(line, false)`.
+export fn trapEntry() align(64) callconv(.naked) noreturn {
+ asm volatile (
+ \\ addi sp, sp, -64
+ \\ sw ra, 0(sp)
+ \\ sw t0, 4(sp)
+ \\ sw t1, 8(sp)
+ \\ sw t2, 12(sp)
+ \\ sw a0, 16(sp)
+ \\ sw a1, 20(sp)
+ \\ sw a2, 24(sp)
+ \\ sw a3, 28(sp)
+ \\ sw a4, 32(sp)
+ \\ sw a5, 36(sp)
+ \\ sw a6, 40(sp)
+ \\ sw a7, 44(sp)
+ \\ sw t3, 48(sp)
+ \\ sw t4, 52(sp)
+ \\ sw t5, 56(sp)
+ \\ sw t6, 60(sp)
+ \\ csrr a0, mcause
+ // Bit 31 set means interrupt, so mcause read as *signed* is negative. `bgez` therefore
+ // branches exactly on "this was an exception", in one instruction and with no scratch
+ // register - which matters here because every scratch register is already spoken for.
+ \\ bgez a0, 1f
+ // mcause[11:0] is the CLIC's interrupt ID. Isolated with a shift pair rather than `andi`:
+ // andi's immediate is 12-bit *signed*, so `andi a0, a0, 0xfff` does not assemble as a
+ // 12-bit mask - it is -1, and would leave the interrupt bit and the level field in place.
+ \\ slli a0, a0, 20
+ \\ srli a0, a0, 20
+ \\ call intrDispatch
+ \\ lw ra, 0(sp)
+ \\ lw t0, 4(sp)
+ \\ lw t1, 8(sp)
+ \\ lw t2, 12(sp)
+ \\ lw a0, 16(sp)
+ \\ lw a1, 20(sp)
+ \\ lw a2, 24(sp)
+ \\ lw a3, 28(sp)
+ \\ lw a4, 32(sp)
+ \\ lw a5, 36(sp)
+ \\ lw a6, 40(sp)
+ \\ lw a7, 44(sp)
+ \\ lw t3, 48(sp)
+ \\ lw t4, 52(sp)
+ \\ lw t5, 56(sp)
+ \\ lw t6, 60(sp)
+ \\ addi sp, sp, 64
+ \\ mret
+ // The exception arm. No restore and no `mret`: `intrFault` is noreturn, because resuming
+ // would re-execute the faulting instruction. The saved registers stay on the stack, which
+ // costs 64 bytes that are never reclaimed and is the correct trade for a path that ends in
+ // a parked core with the numbers printed.
+ \\1:
+ \\ csrr a1, mepc
+ \\ csrr a2, mtval
+ \\ call intrFault
+ );
+}
+
+// ------------------------------------------------------------------------------------ tests
+
+test "the enum's IDs are the offsets of the matrix registers they name" {
+ // The whole of `routeId` rests on `map_reg_addr == base + 4*id`. These four are checked against
+ // the addresses ESP-IDF's own interrupt_core0_reg.h computes, which is an independent path:
+ // IDF wrote the offset as a literal per source, this file multiplies.
+ try std.testing.expectEqual(@as(u32, 0x7c), 4 * @as(u32, @intFromEnum(Source.uart0)));
+ try std.testing.expectEqual(@as(u32, 0xb0), 4 * @as(u32, @intFromEnum(Source.i2c0)));
+ try std.testing.expectEqual(@as(u32, 0xd0), 4 * @as(u32, @intFromEnum(Source.ledc)));
+ try std.testing.expectEqual(@as(u32, 0x1fc), 4 * @as(u32, @intFromEnum(Source.assist_debug)));
+}
+
+test "rev-3-only sources are flagged and the pre-v3 ones are not" {
+ try std.testing.expect(Source.axi_perf_mon.isRev3Only());
+ try std.testing.expect(Source.dma2d_in_ch2.isRev3Only());
+ try std.testing.expect(!Source.assist_debug.isRev3Only());
+ try std.testing.expect(!Source.dma2d_in_ch1.isRev3Only());
+}
+
+test "priority and threshold use different encodings of the same three bits" {
+ // Priority pads low with zeros, threshold pads low with ones. Getting these the same way round
+ // is the mistake this test exists to catch.
+ const priority_byte = @as(u32, 5) << nlbits_shift;
+ const threshold_byte = (@as(u32, 5) << nlbits_shift) | nlbits_pad;
+ try std.testing.expectEqual(@as(u32, 0xa0), priority_byte);
+ try std.testing.expectEqual(@as(u32, 0xbf), threshold_byte);
+ try std.testing.expectEqual(@as(u32, 5), priority_byte >> nlbits_shift);
+ try std.testing.expectEqual(@as(u32, 5), threshold_byte >> nlbits_shift);
+}
+
+test "trigger's low bit, not its value, decides edge versus level" {
+ try std.testing.expect(Trigger.rising_edge.isEdge());
+ try std.testing.expect(Trigger.falling_edge.isEdge());
+ try std.testing.expect(!Trigger.level.isEdge());
+ try std.testing.expect(!Trigger.level_alias.isEdge());
+}
+
+test "the vector table is aligned to a power of two above its own size" {
+ try std.testing.expectEqual(@as(usize, 256), @alignOf(@TypeOf(vector_table)));
+ try std.testing.expect(@sizeOf(@TypeOf(vector_table)) <= 256);
+}
+
+test "mcause's sign bit is what separates an interrupt from an exception" {
+ // The trap entry branches on `bgez mcause`, which is only correct if bit 31 is the interrupt
+ // flag and the value is read signed. Spelled out here because the asm cannot say it.
+ const interrupt_mcause: u32 = 0x8000_0015; // CLIC ID 21 = external line 5
+ const exception_mcause: u32 = 0x0000_0002; // illegal instruction
+ try std.testing.expect(@as(i32, @bitCast(interrupt_mcause)) < 0);
+ try std.testing.expect(@as(i32, @bitCast(exception_mcause)) >= 0);
+ // And the ID extraction the two shifts perform.
+ try std.testing.expectEqual(@as(u32, 21), (interrupt_mcause << 20) >> 20);
+}
+
+test "mtvec's mode bits do not collide with a 64-byte-aligned base" {
+ // The hardware target is `mtvec[31:6] << 6`, so the mode goes in bits the base cannot use -
+ // but only if the base really is 64-byte aligned. This is the arithmetic `init` performs;
+ // whether the *linked* trapEntry satisfies it is a fact about the link, and
+ // `trapEntryAddress()` is how a test on the die checks that, the image being stripped.
+ const aligned_base: u32 = 0x4000_1200;
+ const mtvec = aligned_base | mtvec_mode_clic;
+ try std.testing.expectEqual(aligned_base, (mtvec >> 6) << 6);
+ // A base one instruction short of alignment vectors 60 bytes early, silently.
+ const bad_base: u32 = 0x4000_1204;
+ try std.testing.expect(((bad_base | mtvec_mode_clic) >> 6) << 6 != bad_base);
+}
diff --git a/src/hal/ledc.zig b/src/hal/ledc.zig
new file mode 100644
index 0000000..62eacd8
--- /dev/null
+++ b/src/hal/ledc.zig
@@ -0,0 +1,587 @@
+//! LEDC: the LED PWM controller. Four timers, eight channels, and the first **shadow-register**
+//! peripheral in this HAL.
+//!
+//! Three things make LEDC different from everything else here, and all three are load-bearing.
+//!
+//! **1. Configuration is staged, then committed.** `LEDC_PARA_UP_CHn` (channel) and
+//! `LEDC_TIMERn_PARA_UP` (timer) are write-to-trigger bits: writing 1 copies the staged fields into
+//! the shadow registers the counter and comparators actually use, and the hardware clears the bit
+//! again by itself (`ledc_reg.h:42-47`, `:951-958`). Values written without a commit are visible in
+//! the register file and have no effect on the output. So every mutator here stages, and every
+//! commit is its own store - `commitChannel` / `commitTimer` - exactly as ESP-IDF's
+//! `ledc_ll_ls_channel_update` (ledc_ll.h:435-438) and `ledc_ll_ls_timer_update` (ledc_ll.h:286-290)
+//! do it.
+//!
+//! The commit store is a read-modify-write, and that is deliberate rather than sloppy: the commit
+//! bit shares its word with the staged fields it commits. `LEDC_PARA_UP_CH0` is bit 4 of
+//! `LEDC_CH0_CONF0_REG`, whose other fields are `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV` and `OVF_*`, so
+//! a bare `writeRaw(1 << 4)` would erase the very configuration it was meant to commit. Compare
+//! `systimer.zig`'s `op.write(.{update.is(1)})`, which is a single whole-word store because
+//! `SYSTIMER_UNIT0_OP_REG` contains nothing else. The read-modify-write is safe here for the reason
+//! `mmio.zig` gives: `PARA_UP` is `WT`, it reads back 0, so the read half of the read-modify-write
+//! can never re-trigger an earlier commit. That is the difference between a self-clearing bit and a
+//! write-1-to-clear bit, and it is why LEDC does not need the interrupt-status treatment.
+//!
+//! **2. The divider is fixed point, Q10.8.** `LEDC_CLK_DIV_TIMERn` is an 18-bit field at [22:5]
+//! (`ledc_reg.h:921-928`) holding a divider with 8 fractional bits (`LEDC_LL_FRACTIONAL_BITS`,
+//! ledc_ll.h:30): bits [17:8] are the integer part, bits [7:0] the fraction, so the value 0x4E2
+//! means 1250/256 = 4.8828. The output frequency is
+//!
+//! f_pwm = f_src * 256 / (div * 2^duty_res)
+//!
+//! and `divisor()` below is ESP-IDF's arithmetic for the inverse, transcribed operation for
+//! operation from `esp_driver_ledc/src/ledc.c:468-497` - including the two places where it is
+//! surprising. See its comment.
+//!
+//! **3. On the P4 the clock mux left the peripheral.** `LEDC_CONF_REG.LEDC_APB_CLK_SEL` still exists
+//! in the register map and still documents an encoding (0: APB, 1: RC_FAST, 2: XTAL), and ESP-IDF's
+//! P4 LL never touches it: the real mux is `HP_SYS_CLKRST.PERI_CLK_CTRL22.REG_LEDC_CLK_SRC_SEL`,
+//! with a *different* encoding (0: XTAL, 1: RC_FAST, 2: PLL_DIV) - ledc_ll.h:223-242. Writing the
+//! in-block register would silently do nothing, and reading it back to check would silently agree.
+//! `ClockSource` below is the HP_SYS_CLKRST encoding.
+//!
+//! Gamma fade *ramps* are out of scope, but one gamma register is not optional: the P4 moved
+//! `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` out of `LEDC_CHn_CONF1_REG` - which on this die
+//! holds only `DUTY_START` - and into gamma RAM. A constant duty is therefore a degenerate one-step
+//! fade, and `setDuty` writes that single entry, which is what ESP-IDF's `ledc_duty_config` does for
+//! every plain duty change (ledc.c:263-280).
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const clkrst = @import("clkrst.zig");
+const gpio = @import("gpio.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+/// Eight channels, four timers (`soc_caps.h:385-386`).
+pub const channel_count = 8;
+pub const timer_count = 4;
+
+/// The counter is 20 bits, so the duty resolution is at most 20 (`soc_caps.h:387`). The register
+/// field is five bits wide and will happily accept 21-31; the hardware will not.
+pub const max_duty_resolution = 20;
+
+/// Fractional bits in `LEDC_CLK_DIV_TIMERn` - `LEDC_LL_FRACTIONAL_BITS`, ledc_ll.h:30.
+pub const fractional_bits = 8;
+
+/// The divider must be at least 1.0 and must fit the field: ESP-IDF's `LEDC_IS_DIV_INVALID`
+/// (ledc.c:114) rejects anything `<= LEDC_LL_FRACTIONAL_MAX` or `> LEDC_TIMER_DIV_NUM_MAX`.
+pub const divisor_min: u32 = 1 << fractional_bits;
+pub const divisor_max: u32 = 0x3ffff;
+
+pub const Error = error{
+ /// The requested frequency cannot be reached from this source at this resolution: the divider
+ /// would be below 1.0 (frequency too high) or wider than 18 bits (frequency too low).
+ DividerOutOfRange,
+ DutyResolutionOutOfRange,
+};
+
+// ------------------------------------------------------------------------------------- registers
+
+// Five registers per channel, stride 0x14; two per timer, stride 0x08. Both strides come from a
+// second instance's macro rather than being assumed - see mmio.RegArray.
+const ch_conf0 = mmio.RegArray(regs.LEDC_CH0_CONF0_REG, regs.LEDC_CH1_CONF0_REG, channel_count);
+const ch_hpoint = mmio.RegArray(regs.LEDC_CH0_HPOINT_REG, regs.LEDC_CH1_HPOINT_REG, channel_count);
+const ch_duty = mmio.RegArray(regs.LEDC_CH0_DUTY_REG, regs.LEDC_CH1_DUTY_REG, channel_count);
+const ch_conf1 = mmio.RegArray(regs.LEDC_CH0_CONF1_REG, regs.LEDC_CH1_CONF1_REG, channel_count);
+const ch_duty_r = mmio.RegArray(regs.LEDC_CH0_DUTY_R_REG, regs.LEDC_CH1_DUTY_R_REG, channel_count);
+const ch_gamma_conf = mmio.RegArray(regs.LEDC_CH0_GAMMA_CONF_REG, regs.LEDC_CH1_GAMMA_CONF_REG, channel_count);
+// Gamma RAM: 16 entries per channel, so the per-channel stride is 0x40 and entry 0 is the base.
+const ch_gamma_range0 = mmio.RegArray(regs.LEDC_CH0_GAMMA_RANGE0_REG, regs.LEDC_CH1_GAMMA_RANGE0_REG, channel_count);
+const tim_conf = mmio.RegArray(regs.LEDC_TIMER0_CONF_REG, regs.LEDC_TIMER1_CONF_REG, timer_count);
+const tim_value = mmio.RegArray(regs.LEDC_TIMER0_VALUE_REG, regs.LEDC_TIMER1_VALUE_REG, timer_count);
+
+// Field geometry is taken from instance 0 and reused for every instance, which is only sound if the
+// instances agree; the comptime block below checks the ends of both ranges against instance 0. That
+// is not paranoia about the silicon, it is paranoia about the macro names: `LEDC_CLK_DIV_TIMER0` and
+// `LEDC_TIMER0_DUTY_RES` put the instance number in different places, and picking up
+// `LEDC_TIMER1_DUTY_RES_S` while meaning timer 0's shift is a one-character mistake.
+const timer_sel = Field.of(regs.LEDC_TIMER_SEL_CH0_S, regs.LEDC_TIMER_SEL_CH0_V);
+const sig_out_en = Field.of(regs.LEDC_SIG_OUT_EN_CH0_S, regs.LEDC_SIG_OUT_EN_CH0_V);
+const idle_lv = Field.of(regs.LEDC_IDLE_LV_CH0_S, regs.LEDC_IDLE_LV_CH0_V);
+const ch_para_up = Field.of(regs.LEDC_PARA_UP_CH0_S, regs.LEDC_PARA_UP_CH0_V);
+const hpoint = Field.of(regs.LEDC_HPOINT_CH0_S, regs.LEDC_HPOINT_CH0_V);
+const duty = Field.of(regs.LEDC_DUTY_CH0_S, regs.LEDC_DUTY_CH0_V);
+const duty_r = Field.of(regs.LEDC_DUTY_CH0_R_S, regs.LEDC_DUTY_CH0_R_V);
+const duty_start = Field.of(regs.LEDC_DUTY_START_CH0_S, regs.LEDC_DUTY_START_CH0_V);
+const gamma_entry_num = Field.of(regs.LEDC_CH0_GAMMA_ENTRY_NUM_S, regs.LEDC_CH0_GAMMA_ENTRY_NUM_V);
+const gamma_duty_inc = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_INC_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_INC_V);
+const gamma_duty_cycle = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_CYCLE_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_CYCLE_V);
+const gamma_scale = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_SCALE_S, regs.LEDC_CH0_GAMMA_RANGE0_SCALE_V);
+const gamma_duty_num = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_NUM_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_NUM_V);
+
+const duty_res = Field.of(regs.LEDC_TIMER0_DUTY_RES_S, regs.LEDC_TIMER0_DUTY_RES_V);
+const clk_div = Field.of(regs.LEDC_CLK_DIV_TIMER0_S, regs.LEDC_CLK_DIV_TIMER0_V);
+const tim_pause = Field.of(regs.LEDC_TIMER0_PAUSE_S, regs.LEDC_TIMER0_PAUSE_V);
+const tim_rst = Field.of(regs.LEDC_TIMER0_RST_S, regs.LEDC_TIMER0_RST_V);
+const tim_para_up = Field.of(regs.LEDC_TIMER0_PARA_UP_S, regs.LEDC_TIMER0_PARA_UP_V);
+
+comptime {
+ const same = struct {
+ fn check(comptime what: []const u8, comptime a: Field, comptime b: Field) void {
+ if (a.shift != b.shift or a.width != b.width) @compileError(
+ "the per-instance " ++ what ++ " macros disagree on bit position or width; " ++
+ "this file must index the field per instance instead of reusing instance 0's",
+ );
+ }
+ }.check;
+ // Channels: 1 and 7, the two ends of the range beyond instance 0.
+ same("LEDC_TIMER_SEL_CHn", timer_sel, Field.of(regs.LEDC_TIMER_SEL_CH1_S, regs.LEDC_TIMER_SEL_CH1_V));
+ same("LEDC_TIMER_SEL_CHn", timer_sel, Field.of(regs.LEDC_TIMER_SEL_CH7_S, regs.LEDC_TIMER_SEL_CH7_V));
+ same("LEDC_SIG_OUT_EN_CHn", sig_out_en, Field.of(regs.LEDC_SIG_OUT_EN_CH7_S, regs.LEDC_SIG_OUT_EN_CH7_V));
+ same("LEDC_IDLE_LV_CHn", idle_lv, Field.of(regs.LEDC_IDLE_LV_CH7_S, regs.LEDC_IDLE_LV_CH7_V));
+ same("LEDC_PARA_UP_CHn", ch_para_up, Field.of(regs.LEDC_PARA_UP_CH7_S, regs.LEDC_PARA_UP_CH7_V));
+ same("LEDC_HPOINT_CHn", hpoint, Field.of(regs.LEDC_HPOINT_CH7_S, regs.LEDC_HPOINT_CH7_V));
+ same("LEDC_DUTY_CHn", duty, Field.of(regs.LEDC_DUTY_CH7_S, regs.LEDC_DUTY_CH7_V));
+ same("LEDC_DUTY_START_CHn", duty_start, Field.of(regs.LEDC_DUTY_START_CH7_S, regs.LEDC_DUTY_START_CH7_V));
+ same("LEDC_CHn_GAMMA_ENTRY_NUM", gamma_entry_num, Field.of(regs.LEDC_CH7_GAMMA_ENTRY_NUM_S, regs.LEDC_CH7_GAMMA_ENTRY_NUM_V));
+ same("LEDC_CHn_GAMMA_RANGE0_SCALE", gamma_scale, Field.of(regs.LEDC_CH7_GAMMA_RANGE0_SCALE_S, regs.LEDC_CH7_GAMMA_RANGE0_SCALE_V));
+ // Timers: 1 and 3.
+ same("LEDC_TIMERn_DUTY_RES", duty_res, Field.of(regs.LEDC_TIMER1_DUTY_RES_S, regs.LEDC_TIMER1_DUTY_RES_V));
+ same("LEDC_TIMERn_DUTY_RES", duty_res, Field.of(regs.LEDC_TIMER3_DUTY_RES_S, regs.LEDC_TIMER3_DUTY_RES_V));
+ same("LEDC_CLK_DIV_TIMERn", clk_div, Field.of(regs.LEDC_CLK_DIV_TIMER3_S, regs.LEDC_CLK_DIV_TIMER3_V));
+ same("LEDC_TIMERn_PAUSE", tim_pause, Field.of(regs.LEDC_TIMER3_PAUSE_S, regs.LEDC_TIMER3_PAUSE_V));
+ same("LEDC_TIMERn_RST", tim_rst, Field.of(regs.LEDC_TIMER3_RST_S, regs.LEDC_TIMER3_RST_V));
+ same("LEDC_TIMERn_PARA_UP", tim_para_up, Field.of(regs.LEDC_TIMER3_PARA_UP_S, regs.LEDC_TIMER3_PARA_UP_V));
+
+ // `LEDC_TIMER_DIV_NUM_MAX` (ledc.c:110) is a literal in the driver; it should be the field's
+ // own mask, and if a future die widens the field this is where the two part company.
+ if (divisor_max != clk_div.max()) @compileError(
+ "divisor_max no longer matches LEDC_CLK_DIV_TIMERn's width",
+ );
+ // The eight output signals must be consecutive for `signalIndex` to be arithmetic.
+ if (regs.LEDC_LS_SIG_OUT_PAD_OUT7_IDX - regs.LEDC_LS_SIG_OUT_PAD_OUT0_IDX != channel_count - 1)
+ @compileError("the LEDC output signal indices are not consecutive; signalIndex must be a table");
+}
+
+// ------------------------------------------------------------------------------ clocks and reset
+
+/// LEDC's function clock, in HP_SYS_CLKRST rather than in the peripheral (ledc_ll.h:179, :241).
+/// Shared with RMT's fields, hence the interrupt-masked read-modify-write.
+const peri_clk_ctrl22 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL22_REG);
+const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_LEDC_CLK_SRC_SEL_V);
+const func_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_CLK_EN_S, regs.HP_SYS_CLKRST_REG_LEDC_CLK_EN_V);
+
+/// The four timers' shared source. Encoding from `ledc_ll_set_slow_clk_sel` (ledc_ll.h:223-242) -
+/// *not* the encoding `LEDC_CONF_REG.APB_CLK_SEL` documents, which is a different register on a
+/// different block and is dead on this die.
+pub const ClockSource = enum(u2) {
+ /// 40 MHz on this board (`clk_tree_defs.h:145`).
+ xtal = 0,
+ /// The internal RC oscillator: approximately 17.5 MHz (`clk_tree_defs.h:58`) and not trimmed.
+ /// ESP-IDF calibrates it against XTAL before using it for a divider; there is no calibration
+ /// here, so a frequency computed from `rc_fast_hz_approx` is approximate too.
+ rc_fast = 1,
+ /// PLL_F80M, 80 MHz (`clk_tree_defs.h:168`). Called `LEDC_SLOW_CLK_PLL_DIV` by ESP-IDF.
+ pll_div = 2,
+
+ /// The source frequency to feed `divisor`, or null for RC_FAST, whose real rate has to be
+ /// measured rather than assumed.
+ pub fn hz(self: ClockSource) ?u32 {
+ return switch (self) {
+ .xtal => xtal_hz,
+ .pll_div => pll_div_hz,
+ .rc_fast => null,
+ };
+ }
+};
+
+pub const xtal_hz: u32 = 40_000_000;
+pub const pll_div_hz: u32 = 80_000_000;
+pub const rc_fast_hz_approx: u32 = 17_500_000;
+
+/// Select the timers' source clock. A read-modify-write of a register that also holds RMT's clock
+/// fields, so it runs with interrupts masked, like everything else that touches HP_SYS_CLKRST.
+pub fn setClockSource(src: ClockSource) void {
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ peri_clk_ctrl22.modify(.{clk_src_sel.is(@intFromEnum(src))});
+}
+
+pub fn getClockSource() ClockSource {
+ return @enumFromInt(peri_clk_ctrl22.get(clk_src_sel));
+}
+
+/// LEDC's core ("function") clock gate. Distinct from the APB gate in `clkrst`: the APB clock makes
+/// the registers addressable, this one makes the counters run - and ESP-IDF notes that some LEDC
+/// registers and the gamma RAM need it just to be read or written (ledc.c:433-436).
+pub fn setFunctionClockEnabled(on: bool) void {
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ peri_clk_ctrl22.modify(.{func_clk_en.is(@intFromBool(on))});
+}
+
+/// Bring the peripheral up, in the only order that works: bus clock, reset, function clock, source.
+///
+/// The bus clock first because LEDC is one of the blocks whose APB gate is *off* at power-on
+/// (`hp_sys_clkrst_reg.h:835`, REG_LEDC_APB_CLK_EN default 0), so every register read before this
+/// returns the last value the bus latched. The function clock before any configuration because the
+/// gamma RAM needs it. ESP-IDF deasserts the reset rather than pulsing it (ledc.c:430-431), because
+/// its driver may be attaching to a running LEDC; this pulses, which is the stronger guarantee for a
+/// fresh boot and is measurably safe on this board - pulsing REG_RST_EN_LEDC for 1 ms left the
+/// console untouched and returned LEDC_CH0_CONF0 to 0.
+pub fn init(src: ClockSource) void {
+ clkrst.setClockEnabled(.ledc, true);
+ clkrst.resetPeripheral(.ledc);
+ setFunctionClockEnabled(true);
+ setClockSource(src);
+}
+
+// -------------------------------------------------------------------------------- divider maths
+
+/// ESP-IDF's `ledc_calculate_divisor`, transcribed from `esp_driver_ledc/src/ledc.c:468-497`:
+///
+/// return (((uint64_t) src_clk_freq << LEDC_LL_FRACTIONAL_BITS) + freq_hz * precision / 2)
+/// / (freq_hz * precision);
+///
+/// Result is Q10.8 - see the file comment - and `divisorValid` says whether it fits the field.
+///
+/// Two properties of that C expression are not obvious and are reproduced deliberately, because a
+/// HAL that computed a *better* divider than IDF's would disagree with it on real inputs and there
+/// would be no way to tell which of the two was wrong:
+///
+/// 1. `freq_hz * precision` is `int * uint32_t`, so it is computed in **32 bits and wraps**, and
+/// the wrap is not always harmlessly out of range. Ask for 4097 Hz at 20-bit resolution from the
+/// 40 MHz XTAL: the true product is 2^32 + 2^20, the C code divides by 2^20 instead, and the
+/// answer is 9766 - a *valid* divider, which programs 1.0 Hz. IDF accepts it, because the value
+/// passes its own range check. `%*` here is that wrap, on purpose: reproducing it is what makes
+/// the on-die comparison meaningful, and the numbers above are how a caller can recognise it.
+/// 2. The quotient is `uint64_t` but the return type is `uint32_t`, so it is **truncated**. From a
+/// 40 MHz source at 1 Hz and 1-bit resolution the quotient is 5.12e9 and IDF returns 825032704.
+/// `@truncate` is that truncation.
+///
+/// The one place this cannot follow IDF is `freq_hz * precision == 0`, reachable at exactly 4096 Hz
+/// with 20-bit resolution (2^32, wrapping to zero), where the C code divides by zero. Returning 0 is
+/// a deliberate substitution: it is not a valid divider, so `divisorValid` rejects it and the caller
+/// gets an error instead of undefined behaviour.
+pub fn divisor(src_hz: u32, freq_hz: u32, resolution: u5) u32 {
+ const precision: u32 = @as(u32, 1) << resolution;
+ const den: u32 = freq_hz *% precision;
+ if (den == 0) return 0;
+ const num: u64 = (@as(u64, src_hz) << fractional_bits) + den / 2;
+ return @truncate(num / den);
+}
+
+/// `LEDC_IS_DIV_INVALID`, inverted (ledc.c:114). A divider below 1.0 means the requested frequency
+/// is faster than the source can produce at that resolution.
+pub fn divisorValid(div: u32) bool {
+ return div >= divisor_min and div <= divisor_max;
+}
+
+/// The frequency a given divider and resolution actually produce: `f_src * 256 / (div * 2^res)`,
+/// rounded, and 0 for a divider of 0.
+///
+/// This is `ledc_get_freq`'s arithmetic (ledc.c:1175) with one deliberate difference: the
+/// denominator is computed in 64 bits, so it does not wrap. Nothing compares this against IDF - it
+/// is a convenience for callers checking what they got - and a wrapped denominator here would be a
+/// bug rather than a compatibility requirement.
+pub fn frequencyOf(src_hz: u32, div: u32, resolution: u5) u32 {
+ if (div == 0) return 0;
+ const den: u64 = @as(u64, div) * (@as(u64, 1) << resolution);
+ const num: u64 = (@as(u64, src_hz) << fractional_bits) + den / 2;
+ return @truncate(num / den);
+}
+
+// --------------------------------------------------------------------------------------- timers
+
+/// Stage the divider. `ledc_ll_set_clock_divider`, ledc_ll.h:345-348.
+pub fn setClockDivider(timer: u32, div: u32) void {
+ std.debug.assert(timer < timer_count);
+ tim_conf.at(timer).modify(.{clk_div.is(div)});
+}
+
+pub fn getClockDivider(timer: u32) u32 {
+ std.debug.assert(timer < timer_count);
+ return tim_conf.at(timer).get(clk_div);
+}
+
+/// Stage the duty resolution, in bits. `ledc_ll_set_duty_resolution`, ledc_ll.h:391-394.
+pub fn setDutyResolution(timer: u32, bits: u5) void {
+ std.debug.assert(timer < timer_count);
+ std.debug.assert(bits <= max_duty_resolution);
+ tim_conf.at(timer).modify(.{duty_res.is(bits)});
+}
+
+pub fn getDutyResolution(timer: u32) u5 {
+ std.debug.assert(timer < timer_count);
+ return @intCast(tim_conf.at(timer).get(duty_res));
+}
+
+/// Commit the staged divider and resolution. One store, and the bit clears itself.
+///
+/// ESP-IDF does not wait for it: "we don't wait for the bit gets cleared since it can take quite
+/// long depends on the pwm frequency" (ledc_ll.h:289). Neither does this - a poll here would block
+/// for a whole PWM period, and there is nothing useful to do with the answer.
+pub fn commitTimer(timer: u32) void {
+ std.debug.assert(timer < timer_count);
+ tim_conf.at(timer).modify(.{tim_para_up.is(1)});
+}
+
+/// Reset the timer's counter: assert, deassert (`ledc_ll_timer_rst`, ledc_ll.h:301-305).
+///
+/// Note the reset value of `LEDC_TIMERn_RST` is **1** (ledc_reg.h:936-943), which is one of the
+/// 46.7% of fields whose reset value is not zero, and the reason `configureTimer` finishes by
+/// clearing it: a freshly reset LEDC block holds all four counters at zero and they stay there until
+/// something writes that bit back down.
+pub fn resetTimer(timer: u32) void {
+ std.debug.assert(timer < timer_count);
+ const r = tim_conf.at(timer);
+ r.modify(.{tim_rst.is(1)});
+ r.modify(.{tim_rst.is(0)});
+}
+
+/// Freeze the counter where it is (`ledc_ll_timer_pause`, ledc_ll.h:316-319).
+pub fn pauseTimer(timer: u32) void {
+ std.debug.assert(timer < timer_count);
+ tim_conf.at(timer).modify(.{tim_pause.is(1)});
+}
+
+pub fn resumeTimer(timer: u32) void {
+ std.debug.assert(timer < timer_count);
+ tim_conf.at(timer).modify(.{tim_pause.is(0)});
+}
+
+/// The counter's current value, 20 bits. Reading it is a plain load - no latch handshake, unlike
+/// systimer.
+pub fn timerCount(timer: u32) u32 {
+ std.debug.assert(timer < timer_count);
+ return tim_value.at(timer).raw();
+}
+
+/// Everything a timer needs to produce `freq_hz` at `resolution` bits, in ESP-IDF's order:
+/// divider, resolution, commit, then out of pause and out of reset (`ledc_set_timer_params`,
+/// ledc.c:244-261, followed by ledc.c:816-818).
+///
+/// Returns `DividerOutOfRange` rather than programming a divider the hardware cannot hold. The
+/// caller passes the source frequency because this HAL has no clock tree: `ClockSource.hz()` gives
+/// it for XTAL and PLL_DIV, and RC_FAST has to be measured.
+pub fn configureTimer(timer: u32, opts: struct {
+ src_hz: u32,
+ freq_hz: u32,
+ resolution: u5,
+}) Error!void {
+ std.debug.assert(timer < timer_count);
+ if (opts.resolution == 0 or opts.resolution > max_duty_resolution) return Error.DutyResolutionOutOfRange;
+ const div = divisor(opts.src_hz, opts.freq_hz, opts.resolution);
+ if (!divisorValid(div)) return Error.DividerOutOfRange;
+
+ setClockDivider(timer, div);
+ setDutyResolution(timer, opts.resolution);
+ commitTimer(timer);
+ resumeTimer(timer);
+ resetTimer(timer);
+}
+
+// ------------------------------------------------------------------------------------- channels
+
+/// Which timer drives this channel. Staged; needs `commitChannel`.
+/// `ledc_ll_bind_channel_timer`, ledc_ll.h:697-700.
+pub fn bindTimer(channel: u32, timer: u32) void {
+ std.debug.assert(channel < channel_count and timer < timer_count);
+ ch_conf0.at(channel).modify(.{timer_sel.is(timer)});
+}
+
+pub fn boundTimer(channel: u32) u32 {
+ std.debug.assert(channel < channel_count);
+ return ch_conf0.at(channel).get(timer_sel);
+}
+
+/// Where in the period the output goes high, in counter ticks. Staged.
+/// `ledc_ll_set_hpoint`, ledc_ll.h:450-453.
+pub fn setHpoint(channel: u32, value: u32) void {
+ std.debug.assert(channel < channel_count and value <= hpoint.max());
+ ch_hpoint.at(channel).modify(.{hpoint.is(value)});
+}
+
+/// Stage a duty value, in counter ticks out of `2^resolution`.
+///
+/// Two things happen here that the name does not suggest, and both are ESP-IDF's
+/// (`ledc_ll_set_duty_int_part` ledc_ll.h:480-483, `ledc_duty_config` ledc.c:263-280):
+///
+/// * The register holds duty in **Q21.4** - four fractional bits, used by fades - so the integer
+/// duty is shifted left by 4. `getDuty` shifts back.
+/// * The P4 has no plain-duty path. `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` moved out of
+/// `CHn_CONF1` into gamma RAM, so a constant duty is a one-step fade of scale 0: entry 0 gets
+/// (increase, one cycle, scale 0, one step) and the range count is set to 1. Without that entry
+/// the staged duty is committed and the output does not move.
+///
+/// Staged; needs `commitChannel` (or `start`, which commits).
+pub fn setDuty(channel: u32, value: u32) void {
+ std.debug.assert(channel < channel_count);
+ std.debug.assert(value <= duty.max() >> 4);
+ ch_duty.at(channel).modify(.{duty.is(value << 4)});
+ stageNoFade(channel);
+}
+
+/// The duty the hardware is currently using, from the read-only shadow (`ledc_ll_get_duty`,
+/// ledc_ll.h:495-498). This is the one register that shows whether a commit actually happened - and
+/// it only updates when the timer next overflows, so it is not a synchronous read-back.
+pub fn currentDuty(channel: u32) u32 {
+ std.debug.assert(channel < channel_count);
+ return ch_duty_r.at(channel).get(duty_r) >> 4;
+}
+
+/// Gamma RAM entry 0 as "no fade": one step, one cycle, scale 0, increasing. Exactly the parameters
+/// `ledc_set_duty` passes down (ledc.c:1109-1117) for a constant duty.
+fn stageNoFade(channel: u32) void {
+ // The whole word is being established, and every field in it is being named, so this is one of
+ // the few places `write` is right rather than `modify`.
+ ch_gamma_range0.at(channel).write(.{
+ gamma_duty_inc.is(1),
+ gamma_duty_cycle.is(1),
+ gamma_scale.is(0),
+ gamma_duty_num.is(1),
+ });
+ ch_gamma_conf.at(channel).modify(.{gamma_entry_num.is(1)});
+}
+
+/// The output driver. Staged; needs `commitChannel`.
+/// `ledc_ll_set_sig_out_en`, ledc_ll.h:592-596.
+pub fn setOutputEnabled(channel: u32, on: bool) void {
+ std.debug.assert(channel < channel_count);
+ ch_conf0.at(channel).modify(.{sig_out_en.is(@intFromBool(on))});
+}
+
+/// The level the pad holds while the channel is disabled - and only while it is disabled
+/// (`ledc_reg.h:34-37`: "Valid only when LEDC_SIG_OUT_EN_CHn is 0"). Staged.
+/// `ledc_ll_set_idle_level`, ledc_ll.h:622-626.
+pub fn setIdleLevel(channel: u32, level: u1) void {
+ std.debug.assert(channel < channel_count);
+ ch_conf0.at(channel).modify(.{idle_lv.is(level)});
+}
+
+/// Hand the staged duty to the fade engine. `ledc_ll_set_duty_start`, ledc_ll.h:607-610.
+///
+/// `DUTY_START` lives in `CHn_CONF1`, alone, and is annotated `R/W/SC` - the hardware clears it when
+/// the (here one-step) fade finishes. A read-modify-write is still the right store: the bit is the
+/// only field in the word, but bits 30:0 are reserved and writing them back as read is what IDF's
+/// bitfield assignment does.
+pub fn startFade(channel: u32) void {
+ std.debug.assert(channel < channel_count);
+ ch_conf1.at(channel).modify(.{duty_start.is(1)});
+}
+
+/// Commit the channel's staged fields: `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV`, `HPOINT`,
+/// `DUTY_START`, `OVF_CNT_EN` and the duty (`ledc_reg.h:42-47`).
+///
+/// One deliberate store, never folded into the store that staged the values, matching
+/// `ledc_ll_ls_channel_update` (ledc_ll.h:435-438). It is a read-modify-write because the commit bit
+/// shares its word with the staged fields - see the file comment - and that is safe only because the
+/// bit reads back as 0.
+pub fn commitChannel(channel: u32) void {
+ std.debug.assert(channel < channel_count);
+ ch_conf0.at(channel).modify(.{ch_para_up.is(1)});
+}
+
+/// Start driving: output on, duty handed over, committed. `_ledc_update_duty`, ledc.c:1021-1026.
+pub fn start(channel: u32) void {
+ setOutputEnabled(channel, true);
+ startFade(channel);
+ commitChannel(channel);
+}
+
+/// Stop driving and hold the pad at `idle_level`. `ledc_stop`, ledc.c:1039-1050.
+///
+/// The order is IDF's and it matters: the idle level is staged *before* the output is disabled, so
+/// the two reach the hardware in the same commit and the pad never spends a period at the old idle
+/// level.
+pub fn stop(channel: u32, idle_level: u1) void {
+ setIdleLevel(channel, idle_level);
+ setOutputEnabled(channel, false);
+ commitChannel(channel);
+}
+
+/// A whole channel in one commit: timer, duty, hpoint, idle level, output enable.
+///
+/// This is the one operation here that is not a transcription of an ESP-IDF function - IDF's
+/// `ledc_channel_config` also allocates a driver object, reserves the pin and installs a fade
+/// service - but it is the same register sequence: stage everything, then commit once. One commit
+/// rather than five is the point: the channel changes all at once, at a period boundary, instead of
+/// drifting through four intermediate configurations.
+pub fn configureChannel(channel: u32, opts: struct {
+ timer: u32,
+ duty: u32,
+ hpoint: u32 = 0,
+ idle_level: u1 = 0,
+ output_enabled: bool = true,
+}) void {
+ bindTimer(channel, opts.timer);
+ setHpoint(channel, opts.hpoint);
+ setDuty(channel, opts.duty);
+ setIdleLevel(channel, opts.idle_level);
+ setOutputEnabled(channel, opts.output_enabled);
+ startFade(channel);
+ commitChannel(channel);
+}
+
+// ------------------------------------------------------------------------------------ pin output
+
+/// The GPIO matrix signal index for a channel's output. `ledc_periph_signal[0].sig_out0_idx` is
+/// `LEDC_LS_SIG_OUT_PAD_OUT0_IDX` (esp_hal_ledc/esp32p4/ledc_periph.c:14-18) and the driver adds the
+/// channel number to it (ledc.c:831); the eight indices are consecutive from 126, asserted above.
+pub fn signalIndex(channel: u32) u32 {
+ std.debug.assert(channel < channel_count);
+ return @as(u32, @intCast(regs.LEDC_LS_SIG_OUT_PAD_OUT0_IDX)) + channel;
+}
+
+/// Route a channel's output to a pad through the GPIO matrix. No LEDC register is involved: the
+/// peripheral has no pad of its own, and this is the whole of `ledc_set_pin`'s hardware effect
+/// (ledc.c:823-836, whose `gpio_matrix_output` is func_sel + matrix source + output-enable control,
+/// gpio_hal.c:60-69).
+pub fn attachPin(channel: u32, pin: u8) void {
+ gpio.matrixOut(pin, signalIndex(channel));
+}
+
+// ----------------------------------------------------------------------------------------- tests
+
+test "the divider is Q10.8: integer part in [17:8], fraction in [7:0]" {
+ // 40 MHz XTAL, 1 kHz, 13-bit resolution. 40e6*256/(1000*8192) = 1250 = 0x4E2, i.e. 4 + 226/256
+ // = 4.8828. Checked against ESP-IDF's own expression compiled on the host over a 1,680-point
+ // sweep of (source, frequency, resolution).
+ try std.testing.expectEqual(@as(u32, 1250), divisor(40_000_000, 1_000, 13));
+ try std.testing.expectEqual(@as(u32, 1250 >> 8), 4);
+ try std.testing.expectEqual(@as(u32, 1250 & 0xff), 226);
+ // And back again, to within the rounding the format allows.
+ try std.testing.expectEqual(@as(u32, 1_000), frequencyOf(40_000_000, 1250, 13));
+}
+
+test "divider values for the frequencies the differential harness uses" {
+ try std.testing.expectEqual(@as(u32, 2000), divisor(40_000_000, 5_000, 10));
+ try std.testing.expectEqual(@as(u32, 500), divisor(40_000_000, 20_000, 10));
+ try std.testing.expectEqual(@as(u32, 2083), divisor(40_000_000, 300, 14));
+ // 80 MHz PLL_F80M, same request: exactly twice the divider.
+ try std.testing.expectEqual(@as(u32, 4000), divisor(80_000_000, 5_000, 10));
+}
+
+test "the arithmetic reproduces IDF's overflow and truncation rather than fixing them" {
+ // 32-bit wrap of freq*precision: the true product at 1 MHz / 13 bits is 8_192_000_000, and the
+ // C expression divides by 3_897_032_704 instead, giving 3 where the unwrapped arithmetic would
+ // give 1. Neither is a usable divider - both are below 1.0, so `divisorValid` rejects them the
+ // way `LEDC_IS_DIV_INVALID` does - but the *value* has to be IDF's, or a caller comparing the
+ // two implementations sees a difference that is really just two different roundings.
+ try std.testing.expectEqual(@as(u32, 1_000_000 *% (@as(u32, 1) << 13)), 3_897_032_704);
+ try std.testing.expectEqual(@as(u32, 3), divisor(40_000_000, 1_000_000, 13));
+ try std.testing.expect(!divisorValid(divisor(40_000_000, 1_000_000, 13)));
+ // u64 quotient truncated to u32, exactly as the C return type does.
+ try std.testing.expectEqual(@as(u32, 825_032_704), divisor(40_000_000, 1, 1));
+ // The one input where IDF divides by zero: 4096 * 2^20 == 2^32.
+ try std.testing.expectEqual(@as(u32, 0), divisor(40_000_000, 4096, 20));
+ try std.testing.expect(!divisorValid(divisor(40_000_000, 4096, 20)));
+ // The wrap that is *not* self-limiting: 4097 Hz at 20 bits gives a divider IDF's own range check
+ // accepts, and it programs 1.0 Hz. Reproduced rather than corrected, because the point of the
+ // differential test is to be wrong in the same way IDF is or not at all.
+ try std.testing.expectEqual(@as(u32, 9766), divisor(40_000_000, 4097, 20));
+ try std.testing.expect(divisorValid(9766));
+ try std.testing.expectEqual(@as(u32, 1), frequencyOf(40_000_000, 9766, 20));
+}
+
+test "validity is the field's range, not the whole u32" {
+ try std.testing.expect(!divisorValid(0xff)); // below 1.0
+ try std.testing.expect(divisorValid(0x100)); // exactly 1.0
+ try std.testing.expect(divisorValid(0x3ffff));
+ try std.testing.expect(!divisorValid(0x40000));
+ // 40 MHz cannot make 5 kHz at 13 bits: that needs a divider of 0.98.
+ try std.testing.expect(!divisorValid(divisor(40_000_000, 5_000, 13)));
+}
diff --git a/src/hal/rwdt.zig b/src/hal/rwdt.zig
new file mode 100644
index 0000000..5002400
--- /dev/null
+++ b/src/hal/rwdt.zig
@@ -0,0 +1,123 @@
+//! The RTC watchdog (RWDT) and the super watchdog (SWD), in the always-on LP domain.
+//!
+//! This module exists because of a bug that had been in every application in this project since the
+//! first one, and was invisible for a simple reason: nothing had ever run for more than eight
+//! seconds.
+//!
+//! The second-stage bootloader arms the RTC watchdog to cover the handover to the application, and
+//! expects the application to take it over - ESP-IDF disables it in `esp_system`'s startup, which a
+//! bare image never runs. So the board resets, and the console says so if anyone looks:
+//!
+//! MARK ZIG_P4_ALIVE beat=8 gpio20 high=1 low=0
+//! rst:0x10 (CHIP_LP_WDT_RESET),boot:0x30f (SPI_FAST_FLASH_BOOT)
+//!
+//! Every demo, every example and every hardware test in this repo had been silently rebooting on a
+//! roughly ten-second cycle. It surfaced only when the differential harness grew past 26 cases and
+//! the run stopped fitting inside one watchdog period - which first looked like "the UART suite
+//! crashes the board", and was not.
+//!
+//! Two watchdogs live here and both have to be dealt with:
+//!
+//! * **RWDT**, the RTC watchdog proper, in `LP_WDT_CONFIG0_REG`. Write-protected.
+//! * **SWD**, the super watchdog, a separate always-on timer whose job is to catch a system that
+//! has stopped feeding everything else. It has its own key and its own register, and the
+//! bootloader leaves it auto-feeding (`bootloader_super_wdt_auto_feed`); an application that
+//! disables RWDT and forgets SWD gets a longer fuse rather than no fuse.
+//!
+//! The write-protect scheme is the same as the timer groups': the key register's *reset value* is
+//! the unlock key, so writing anything else locks it. Writes to a locked register are dropped
+//! silently - no fault, no status bit - which is why `disable()` verifies afterwards and returns
+//! whether it took.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+const config0 = Reg.at(regs.LP_WDT_CONFIG0_REG);
+const wprotect = Reg.at(regs.LP_WDT_WPROTECT_REG);
+const swd_config = Reg.at(regs.LP_WDT_SWD_CONFIG_REG);
+const swd_wprotect = Reg.at(regs.LP_WDT_SWD_WPROTECT_REG);
+
+const wdt_en = Field.of(regs.LP_WDT_WDT_EN_S, regs.LP_WDT_WDT_EN_V);
+/// Flash-boot mode runs the watchdog independently of `wdt_en`, which is how the bootloader keeps
+/// the fuse lit across the handover. Clearing `wdt_en` alone leaves this armed.
+const flashboot_en = Field.of(regs.LP_WDT_WDT_FLASHBOOT_MOD_EN_S, regs.LP_WDT_WDT_FLASHBOOT_MOD_EN_V);
+const swd_disable = Field.of(regs.LP_WDT_SWD_DISABLE_S, regs.LP_WDT_SWD_DISABLE_V);
+const swd_auto_feed = Field.of(regs.LP_WDT_SWD_AUTO_FEED_EN_S, regs.LP_WDT_SWD_AUTO_FEED_EN_V);
+const swd_feed = Field.of(regs.LP_WDT_SWD_FEED_S, regs.LP_WDT_SWD_FEED_V);
+const feed_reg = Reg.at(regs.LP_WDT_FEED_REG);
+const feed_bit = Field.of(regs.LP_WDT_FEED_S, regs.LP_WDT_FEED_V);
+
+/// The unlock key for both blocks, which is also each key register's reset value: "if the register
+/// contains a different value than its reset value, write protection is enabled"
+/// (lp_wdt_reg.h). `LP_WDT_WKEY_VALUE` and `LP_WDT_SWD_WKEY_VALUE` in
+/// esp_hal_wdt/esp32p4/include/hal/lpwdt_ll.h:25,27 are both this number.
+const wkey: u32 = 0x50D8_3AA1;
+
+/// Unlocked access to the RTC watchdog. `defer guard.release()` re-locks.
+pub const Guard = struct {
+ pub inline fn release(_: Guard) void {
+ // Anything that is not the key locks it. ESP-IDF writes 0 (lpwdt_ll.h), so this does too:
+ // it keeps the register comparable against IDF's in a differential test.
+ wprotect.writeRaw(0);
+ }
+};
+
+pub inline fn unlock() Guard {
+ wprotect.writeRaw(wkey);
+ return .{};
+}
+
+/// Turn the RTC watchdog off, and stop the super watchdog behind it.
+///
+/// Returns false if the write did not take, which means the key was wrong: a protected register
+/// swallows writes without complaint, so the only way to know is to read back.
+pub fn disable() bool {
+ {
+ const guard = unlock();
+ defer guard.release();
+ // Both bits, in one store: clearing `wdt_en` while leaving flash-boot mode armed is the
+ // half-fix that still reboots.
+ config0.modify(.{ wdt_en.is(0), flashboot_en.is(0) });
+ }
+
+ // The super watchdog is a separate block with its own key.
+ swd_wprotect.writeRaw(wkey);
+ swd_config.modify(.{ swd_disable.is(1), swd_auto_feed.is(0) });
+ swd_wprotect.writeRaw(0);
+
+ return config0.get(wdt_en) == 0 and config0.get(flashboot_en) == 0 and swd_config.get(swd_disable) == 1;
+}
+
+/// Feed the RTC watchdog instead of disabling it, for an application that would rather keep the
+/// protection.
+///
+/// The counter is fed through its own register, `LP_WDT_FEED_REG`, not through anything in
+/// CONFIG0 - ESP-IDF's `lpwdt_ll_feed` writes `hw->feed.feed = 1`. An earlier version of this
+/// function read a CONFIG0 field and wrote the same value back, which is a pure no-op: the register
+/// ended with the bits it started with and the counter kept running. An application that took this
+/// module's own advice - keep the protection, feed it - would have been reset about ten seconds
+/// later with nothing on the console, which is the exact failure this file exists to document. The
+/// differential harness could not have caught it either, because a no-op leaves the register
+/// bit-identical.
+pub fn feed() void {
+ const guard = unlock();
+ defer guard.release();
+ feed_reg.write(.{feed_bit.is(1)});
+}
+
+/// Feed the super watchdog once. Independent of RWDT and of its own auto-feed setting.
+pub fn feedSuper() void {
+ swd_wprotect.writeRaw(wkey);
+ swd_config.modify(.{swd_feed.is(1)});
+ swd_wprotect.writeRaw(0);
+}
+
+/// Whether either watchdog is still armed - worth printing once at startup, because the symptom of
+/// getting this wrong is a reset ten seconds later with no other clue.
+pub fn armed() bool {
+ return config0.get(wdt_en) == 1 or config0.get(flashboot_en) == 1 or swd_config.get(swd_disable) == 0;
+}
diff --git a/src/hal/sdmmc.zig b/src/hal/sdmmc.zig
new file mode 100644
index 0000000..0bebeb0
--- /dev/null
+++ b/src/hal/sdmmc.zig
@@ -0,0 +1,2002 @@
+//! The SDMMC host controller, driven as an **SDIO host**.
+//!
+//! There is no SD card on this board. Slot 1 of the P4's SDMMC controller goes to an ESP32-C6
+//! running ESP-Hosted coprocessor firmware, which presents itself as a 4-bit SDIO device: CLK 18,
+//! CMD 19, D0-D3 = 14/15/16/17. So this file implements CMD0/CMD5/CMD3/CMD7 and then CMD52/CMD53,
+//! and nothing above them. SD memory cards, SPI mode, CSD/CID decoding and block devices are
+//! deliberately absent - they are a different problem that happens to share a peripheral.
+//!
+//! The controller is a Synopsys DesignWare mobile-storage host. Three of its properties decide the
+//! shape of everything below.
+//!
+//! **The card clock is not the register clock.** CLKDIV, CLKSRC and CLKENA are written on the bus
+//! side and do not reach the card-interface unit until a *clock update command* is issued: a write
+//! to the CMD register with `update_clk_reg` and `start_command` set, which sends nothing to the
+//! card (`sdmmc_reg.h:440-456`, and ESP-IDF's `sd_host_slot_clock_update_command`,
+//! `sd_host_sdmmc.c:896-912`). A driver that programmes a divider and moves on has changed
+//! nothing. Three such commands are needed to change frequency safely - clock off, reprogramme,
+//! clock on - and that is what `setBusClock` does.
+//!
+//! **The command register is a single word, and `start_command` is bit 31 of it.** Every attribute
+//! of a command - index, whether a response is expected, whether its CRC is checked, whether data
+//! follows and in which direction - is a field of the same word, and writing that word with bit 31
+//! set launches the command. So the interesting part of "send CMD52" is an encoding, not a
+//! sequence, and `commandWord` is a pure function of the request. It is host-tested, and the
+//! oracle compares the words it produces against words built through ESP-IDF's own
+//! `sdmmc_hw_cmd_t` bitfields.
+//!
+//! **Data moves by internal DMA over descriptors in memory, and the P4 caches that memory.**
+//! `soc_caps.h:185` sets SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE, so L2MEM - where every static in this
+//! image lives - is reached by the CPU through the L1 data cache while the IDMAC reaches it
+//! directly. See the "Cache" section below for the resolution; it is the one place in this file
+//! where the right answer is not visible in any register header.
+//!
+//! Nothing here has been run on hardware by the author of this file. What is claimed is that the
+//! register arithmetic and the command encodings match ESP-IDF's at the cited lines, that
+//! `src/oracle/sdmmc_cases.zig` compares the two on the die, and that the configuration `init`
+//! leaves behind reproduces a dump taken from a working ESP-IDF image on this board.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const gpio = @import("gpio.zig");
+const clkrst = @import("clkrst.zig");
+const intr = @import("intr.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+pub const Error = error{ Timeout, CrcError, ResponseError, NotSupported, Busy };
+
+// ------------------------------------------------------------------------------- registers
+//
+// One instance, at DR_REG_SDHOST_BASE = DR_REG_SDMMC_BASE = 0x50083000 (`reg_base.h:44`, `:204`,
+// and `esp32p4.peripherals.ld:41` agrees). The macros are spelled SDHOST_*, the peripheral is
+// spelled SDMMC, and both names are ESP-IDF's.
+
+const ctrl = Reg.at(regs.SDHOST_CTRL_REG);
+const clkdiv = Reg.at(regs.SDHOST_CLKDIV_REG);
+const clksrc = Reg.at(regs.SDHOST_CLKSRC_REG);
+const clkena = Reg.at(regs.SDHOST_CLKENA_REG);
+const tmout = Reg.at(regs.SDHOST_TMOUT_REG);
+const ctype = Reg.at(regs.SDHOST_CTYPE_REG);
+const blksiz = Reg.at(regs.SDHOST_BLKSIZ_REG);
+const bytcnt = Reg.at(regs.SDHOST_BYTCNT_REG);
+const intmask = Reg.at(regs.SDHOST_INTMASK_REG);
+const cmdarg = Reg.at(regs.SDHOST_CMDARG_REG);
+const cmd = Reg.at(regs.SDHOST_CMD_REG);
+const resp0 = Reg.at(regs.SDHOST_RESP0_REG);
+const rintsts = Reg.at(regs.SDHOST_RINTSTS_REG);
+/// The *masked* status: RINTSTS gated by INTMASK, and the only word the controller's interrupt
+/// output looks at. ESP-IDF's `sdmmc_ll_get_intr_status` reads this one and not RINTSTS
+/// (`sdmmc_ll.h:841-844`), which is exactly why INTMASK decides what reaches the CLIC while
+/// RINTSTS stays readable for the polling path.
+const mintsts = Reg.at(regs.SDHOST_MINTSTS_REG);
+const status = Reg.at(regs.SDHOST_STATUS_REG);
+const fifoth = Reg.at(regs.SDHOST_FIFOTH_REG);
+const bmod = Reg.at(regs.SDHOST_BMOD_REG);
+const pldmnd = Reg.at(regs.SDHOST_PLDMND_REG);
+const dbaddr = Reg.at(regs.SDHOST_DBADDR_REG);
+const idsts = Reg.at(regs.SDHOST_IDSTS_REG);
+const idinten = Reg.at(regs.SDHOST_IDINTEN_REG);
+
+// CTRL fields. Two of them - `dma_enable` at bit 5 and `use_internal_dma` at bit 25 - have no
+// `_S`/`_V` macro pair in `sdmmc_reg.h` at all: that header documents CTRL as bits 0,1,2,4,6..11
+// and stops. They are real, they are in the measured working dump (`ctrl=0x02000030`), and
+// `sdmmc_struct.h:76` and `:135` name them at exactly those positions. This is the same situation
+// as the IO MUX pull bits in `hal/gpio.zig`, and the same remedy: `Field.bit` with the struct
+// header cited, because the struct header is ESP-IDF's definition of the layout even where the
+// macro header is incomplete.
+const controller_reset = Field.of(regs.SDHOST_CONTROLLER_RESET_S, regs.SDHOST_CONTROLLER_RESET_V);
+const fifo_reset = Field.of(regs.SDHOST_FIFO_RESET_S, regs.SDHOST_FIFO_RESET_V);
+const dma_reset = Field.of(regs.SDHOST_DMA_RESET_S, regs.SDHOST_DMA_RESET_V);
+const int_enable = Field.of(regs.SDHOST_INT_ENABLE_S, regs.SDHOST_INT_ENABLE_V);
+/// `sdmmc_struct.h:76` - `uint32_t dma_enable:1;` immediately after `int_enable:1` at bit 4.
+const dma_enable = Field.bit(5);
+/// `sdmmc_struct.h:135` - after `reserved2:4`, `card_voltage_a:4`, `card_voltage_b:4` and
+/// `enable_od_pullup:1`, i.e. bit 25. `sdmmc_ll_enable_dma` (`sdmmc_ll.h:812-818`) is the only
+/// writer, and the working dump's `ctrl=0x02000030` has exactly this bit plus 4 and 5.
+const use_internal_dma = Field.bit(25);
+
+const clk_divider0 = Field.of(regs.SDHOST_CLK_DIVIDER0_S, regs.SDHOST_CLK_DIVIDER0_V);
+const clk_divider1 = Field.of(regs.SDHOST_CLK_DIVIDER1_S, regs.SDHOST_CLK_DIVIDER1_V);
+// CLKSRC is documented as one 4-bit field, two bits per card ("bit[1:0] are assigned for card 0,
+// bit[3:2] are assigned for card 1", `sdmmc_reg.h:166-179`). `sdmmc_struct.h:191-192` splits it
+// into `card0:2` and `card1:2`, which is the shape a driver wants; there are no macros for the
+// halves, so the two sub-fields are spelled out with that citation.
+const clksrc_card0 = Field.of(0, 0x3);
+const clksrc_card1 = Field.of(2, 0x3);
+const cclk_enable = Field.of(regs.SDHOST_CCLK_ENABLE_S, regs.SDHOST_CCLK_ENABLE_V);
+const lp_enable = Field.of(regs.SDHOST_LP_ENABLE_S, regs.SDHOST_LP_ENABLE_V);
+const response_timeout = Field.of(regs.SDHOST_RESPONSE_TIMEOUT_S, regs.SDHOST_RESPONSE_TIMEOUT_V);
+const data_timeout = Field.of(regs.SDHOST_DATA_TIMEOUT_S, regs.SDHOST_DATA_TIMEOUT_V);
+const card_width4 = Field.of(regs.SDHOST_CARD_WIDTH4_S, regs.SDHOST_CARD_WIDTH4_V);
+const card_width8 = Field.of(regs.SDHOST_CARD_WIDTH8_S, regs.SDHOST_CARD_WIDTH8_V);
+const block_size = Field.of(regs.SDHOST_BLOCK_SIZE_S, regs.SDHOST_BLOCK_SIZE_V);
+const byte_count = Field.of(regs.SDHOST_BYTE_COUNT_S, regs.SDHOST_BYTE_COUNT_V);
+const int_mask = Field.of(regs.SDHOST_INT_MASK_S, regs.SDHOST_INT_MASK_V);
+const sdio_int_mask = Field.of(regs.SDHOST_SDIO_INT_MASK_S, regs.SDHOST_SDIO_INT_MASK_V);
+const data_busy = Field.of(regs.SDHOST_DATA_BUSY_S, regs.SDHOST_DATA_BUSY_V);
+const tx_wmark = Field.of(regs.SDHOST_TX_WMARK_S, regs.SDHOST_TX_WMARK_V);
+const rx_wmark = Field.of(regs.SDHOST_RX_WMARK_S, regs.SDHOST_RX_WMARK_V);
+const dma_msize = Field.of(regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_S, regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_V);
+const bmod_swr = Field.of(regs.SDHOST_BMOD_SWR_S, regs.SDHOST_BMOD_SWR_V);
+const bmod_fb = Field.of(regs.SDHOST_BMOD_FB_S, regs.SDHOST_BMOD_FB_V);
+const bmod_de = Field.of(regs.SDHOST_BMOD_DE_S, regs.SDHOST_BMOD_DE_V);
+const idinten_ti = Field.of(regs.SDHOST_IDINTEN_TI_S, regs.SDHOST_IDINTEN_TI_V);
+const idinten_ri = Field.of(regs.SDHOST_IDINTEN_RI_S, regs.SDHOST_IDINTEN_RI_V);
+const idinten_ni = Field.of(regs.SDHOST_IDINTEN_NI_S, regs.SDHOST_IDINTEN_NI_V);
+
+// The host-side clock generator, which is *not* in the SDMMC block: the P4 moved it into
+// HP_SYS_CLKRST, and it is the first of two divider stages (this one, then CLKDIV inside the
+// controller). `sdmmc_ll.h:227-228` for the source mux and gate, `:244-258` for the divider,
+// `:305-315` for the sampling/driving phase clocks.
+const peri_clk_ctrl01 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL01_REG);
+const peri_clk_ctrl02 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL02_REG);
+
+const sdio_hs_mode = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_S, regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_V);
+const sdio_ls_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_V);
+const sdio_ls_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_V);
+const sdio_ls_clk_edge_cfg_update = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_V);
+const sdio_ls_clk_edge_l = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_V);
+const sdio_ls_clk_edge_h = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_V);
+const sdio_ls_clk_edge_n = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_V);
+const sdio_ls_slf_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_V);
+const sdio_ls_drv_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_V);
+const sdio_ls_sam_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_V);
+const sdio_ls_slf_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_V);
+const sdio_ls_drv_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_V);
+const sdio_ls_sam_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_V);
+
+// --------------------------------------------------------------------------------- interrupts
+//
+// RINTSTS / INTMASK share one 16-bit layout plus a 2-bit per-card SDIO field at [17:16].
+// `sdmmc_ll.h:35-53` names every bit; the numbers below are those, not a re-derivation.
+
+pub const Event = struct {
+ pub const cd: u32 = 1 << 0; // card detect
+ pub const re: u32 = 1 << 1; // response error
+ pub const cmd_done: u32 = 1 << 2;
+ pub const dto: u32 = 1 << 3; // data transfer over
+ pub const txdr: u32 = 1 << 4;
+ pub const rxdr: u32 = 1 << 5;
+ pub const rcrc: u32 = 1 << 6; // response CRC error
+ pub const dcrc: u32 = 1 << 7; // data CRC error
+ pub const rto: u32 = 1 << 8; // response timeout
+ pub const drto: u32 = 1 << 9; // data read timeout
+ pub const hto: u32 = 1 << 10; // data starvation by host timeout
+ pub const frun: u32 = 1 << 11; // FIFO under/overrun
+ pub const hle: u32 = 1 << 12; // hardware locked write error
+ pub const sbe: u32 = 1 << 13; // RX start-bit error
+ pub const acd: u32 = 1 << 14; // auto command done
+ pub const ebe: u32 = 1 << 15; // end-bit error
+ pub const io_slot0: u32 = 1 << 16;
+ pub const io_slot1: u32 = 1 << 17;
+
+ /// What `sdmmc_ll.h:64-69` (SDMMC_LL_EVENT_DEFAULT) enables at init. Kept exactly as ESP-IDF
+ /// spells it, because the oracle compares against it; what this driver actually unmasks is
+ /// `armed`, below.
+ pub const default: u32 = cd | re | cmd_done | dto | rcrc | dcrc | rto | drto | hto | hle | sbe | ebe;
+
+ /// `default` without card detect, and the only mask `configureInterrupts` ever writes.
+ ///
+ /// Bit 0 has to go, and this is not a preference. There is no card-detect pin on this board:
+ /// `configurePins` ties the signal to a matrix constant 0 ("card present"), and the
+ /// transition it makes while doing so *latches* RINTSTS.cd. RINTSTS is a sticky
+ /// write-1-to-clear register and nothing in the command path clears bit 0 - `sendCommand`
+ /// deliberately writes `default & ~cd` so as not to disturb asynchronous events. So with cd
+ /// unmasked, the controller's single output line into the CLIC is asserted from bring-up
+ /// onwards and never deasserts, and anyone who enables that CLIC line takes an interrupt
+ /// storm that no handler can end. Found by RxPath on CLIC line 21; the fix belongs here
+ /// rather than in the handler, because a level output that nothing can lower is this file's
+ /// bug.
+ ///
+ /// The two SDIO card-interrupt bits are absent from both masks: `setSlaveInterruptEnabled`
+ /// turns the one for this slot on when somebody is prepared to service it.
+ pub const armed: u32 = default & ~cd;
+
+ /// Anything in here means the command failed. `sdmmc_ll.h:71-77` calls the superset
+ /// SDMMC_LL_SD_EVENT_MASK; this is the error half of it.
+ pub const command_errors: u32 = re | rcrc | rto | hle;
+ pub const data_errors: u32 = dcrc | drto | hto | frun | sbe | ebe;
+};
+
+/// The IDMAC's five reportable events - TI, RI, FBE, DU, CES - as one mask. `sdmmc_ll.h:83`
+/// SDMMC_LL_EVENT_DMA_MASK.
+const idsts_event_mask: u32 = 0x1f;
+
+/// The CLIC source this controller raises, for a caller that wants to be woken rather than to
+/// poll. Registering a handler is `hal.intr`'s job and not this file's: see the note on
+/// `slaveInterruptPending`.
+pub const interrupt_source = intr.Source.sdio_host;
+
+// -------------------------------------------------------------------------------------- cache
+//
+// The IDMAC reads its descriptors and its data buffer straight out of L2MEM. The CPU reaches the
+// same L2MEM through the L1 data cache (`soc_caps.h:185`, SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE), and
+// that cache is write-back: `esp_cache_msync(..., DIR_C2M)` exists precisely because a store the
+// CPU has made may still be sitting in a dirty line when the DMA engine reads memory.
+//
+// ESP-IDF offers two ways out and uses both. `sd_trans_sdmmc.c:135-139` writes descriptors through
+// the normal address and calls `esp_cache_msync` after every one. `gdma_link.c:100-118` does it
+// the other way: one write-back-and-invalidate when the region is created, and from then on every
+// CPU access goes through the non-cacheable alias at `addr + 0x40000000`
+// (`hal/cache_ll.h:27` CACHE_LL_L2MEM_NON_CACHE_ADDR, `soc/ext_mem_defs.h:68`).
+//
+// **This file takes the second route.** It is the cheaper one - no cache call in the transfer
+// path - and it is the only one that stays correct without a cache HAL this project does not have.
+// The one-time write-back-and-invalidate is still required, and skipping it is a real bug rather
+// than a theoretical one: `_start` clears .bss with ordinary stores (`src/main.zig:85-92`), so
+// every word of the DMA region below starts life as a *dirty* cache line full of zeros. Nothing
+// says when those lines are evicted; if one is written back after a descriptor has been prepared
+// through the alias, the descriptor becomes zero and the IDMAC stalls on an unowned descriptor.
+// `gdma_link.c:107-112` does exactly this call for exactly this reason.
+//
+// The two ROM entry points are addressed directly rather than declared `extern`, because the
+// generated linker script provides only `ets_printf` and `ets_delay_us`. The addresses are
+// ESP-IDF's, from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld:186` and `:190` - the hw_ver1
+// file, which is the one that matches this die. (If they move into the linker script beside the
+// other two, these two lines become `extern fn` and nothing else changes.)
+
+/// `soc/ext_mem_defs.h:68` SOC_NON_CACHEABLE_OFFSET.
+pub const non_cacheable_offset: u32 = 0x4000_0000;
+
+/// `cache_ll_l1_dcache_get_line_size` reports this on the P4, and `sdmmc_struct.h:36-38` states it
+/// in prose: "On P4, L1 Cache alignment is 64B".
+pub const cache_line: u32 = 64;
+
+/// `rom/cache.h:230` - CACHE_MAP_L1_DCACHE is BIT(4).
+const cache_map_l1_dcache: u32 = 1 << 4;
+
+const romCacheWriteBackAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int =
+ @ptrFromInt(0x4fc0_03f4);
+const romCacheInvalidateAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int =
+ @ptrFromInt(0x4fc0_03e4);
+
+// ------------------------------------------------------------------------------- DMA descriptor
+
+/// One IDMAC descriptor, exactly as the hardware reads it: `sdmmc_struct.h:13-41`.
+///
+/// ESP-IDF's `sdmmc_desc_t` is 64 bytes, not 16, and its own comment says why and when not to:
+/// "These `reserved[12]` are for cache alignment... For those who want to access the DMA
+/// descriptor in a non-cacheable way, you can consider remove these `reserved[12]` bytes"
+/// (`sdmmc_struct.h:35-39`). That is this file, so the padding is gone and the descriptor is the
+/// 16 bytes the IDMAC actually fetches.
+pub const Descriptor = extern struct {
+ flags: u32,
+ /// [12:0] buffer1_size, [25:13] buffer2_size.
+ sizes: u32,
+ buffer1: u32,
+ /// Also `buffer2_ptr`; which one it is depends on `second_address_chained`.
+ next: u32,
+
+ pub const disable_int_on_completion: u32 = 1 << 1;
+ pub const last_descriptor: u32 = 1 << 2;
+ pub const first_descriptor: u32 = 1 << 3;
+ pub const second_address_chained: u32 = 1 << 4;
+ pub const end_of_ring: u32 = 1 << 5;
+ pub const card_error_summary: u32 = 1 << 30;
+ pub const owned_by_idmac: u32 = 1 << 31;
+
+ /// `sdmmc_struct.h:43` SDMMC_DMA_MAX_BUF_LEN. `buffer1_size` is 13 bits wide, so 8191 would
+ /// fit; ESP-IDF splits at 4096 and so does the bound below.
+ pub const max_buffer_len: u32 = 4096;
+};
+
+/// Bytes of L2MEM this driver owns, and the whole of its dynamic memory: there is no allocator
+/// here and no allocation anywhere in the transfer path.
+///
+/// 2 KiB of payload is chosen against what sits above: ESP-Hosted's SDIO transport moves at most
+/// one 1600-byte frame plus its 12-byte header per CMD53, and the largest single command this
+/// driver can express in block mode is 4 blocks of 512. Anything larger is split across commands
+/// by `transferChunked`, which is correct for both addressing modes, so the number is a
+/// speed/footprint trade and not a limit.
+pub const bounce_len: u32 = 2048;
+
+/// Descriptor and bounce buffer in one cache-line-aligned region, so the one-time maintenance call
+/// is one call over one range whose base and length are both multiples of 64.
+const DmaRegion = extern struct {
+ desc: Descriptor,
+ _pad: [cache_line - @sizeOf(Descriptor)]u8,
+ buf: [bounce_len]u8,
+};
+
+comptime {
+ std.debug.assert(@sizeOf(Descriptor) == 16);
+ std.debug.assert(@sizeOf(DmaRegion) % cache_line == 0);
+ // One descriptor is enough only while the bounce buffer fits in one. If `bounce_len` ever
+ // grows past 4096 this has to become a ring, and this line is what will say so.
+ std.debug.assert(bounce_len <= Descriptor.max_buffer_len);
+}
+
+/// 2112 bytes: 16 of descriptor, 48 of padding to a cache line, 2048 of payload.
+var dma: DmaRegion align(cache_line) = std.mem.zeroes(DmaRegion);
+
+/// Addresses are `usize` rather than `u32` all the way to the register write. On this target the
+/// two are the same type; on the host, where the arithmetic in these helpers is unit-tested,
+/// `@intCast` of a real 64-bit address would panic before the test could check anything.
+inline fn cachedAddr(p: *const anyopaque) usize {
+ return @intFromPtr(p);
+}
+
+/// The address the *CPU* must use for anything in the DMA region. The hardware gets the cached
+/// address - that is not an inconsistency, it is what ESP-IDF does: `gdma_link.c:268-273` hands
+/// `list->items` to the peripheral and `:159` writes through `list->items_nc`. The alias exists to
+/// change how the CPU's loads and stores are treated, and a bus master is not the CPU.
+inline fn uncachedAddr(p: *const anyopaque) usize {
+ return cachedAddr(p) +% @as(usize, non_cacheable_offset);
+}
+
+/// An address as the 32-bit register field the hardware reads it through.
+inline fn busAddr(p: *const anyopaque) u32 {
+ return @intCast(cachedAddr(p));
+}
+
+inline fn descNc() *volatile Descriptor {
+ return @ptrFromInt(uncachedAddr(&dma.desc));
+}
+
+inline fn bufNc() [*]volatile u8 {
+ return @ptrFromInt(uncachedAddr(&dma.buf));
+}
+
+/// Write back and invalidate the DMA region once, so that no dirty line from `_start`'s .bss clear
+/// can later land on top of what the alias writes. After this, the cached alias of this region is
+/// never touched again by anything in this file.
+fn syncDmaRegionOnce() void {
+ const base = busAddr(&dma);
+ const len: u32 = @sizeOf(DmaRegion);
+ _ = romCacheWriteBackAddr(cache_map_l1_dcache, base, len);
+ _ = romCacheInvalidateAddr(cache_map_l1_dcache, base, len);
+}
+
+// -------------------------------------------------------------------------------------- timing
+//
+// Every wait in this file is bounded, and bounded in time rather than in loop iterations: a spin
+// count is a different number on every optimize level, and this board has no debugger, so a wait
+// that never returns is indistinguishable from a crash.
+//
+// The timebase is the RISC-V `cycle` CSR, the unprivileged shadow of `mcycle`, which is what
+// ESP-IDF itself reads on this part (`rv_utils.h`, because SOC_CPU_HAS_CSR_PC is not defined for
+// the P4) and what `src/soc.zig:116-131` already uses. It is deliberately *not* `hal.systimer`:
+// systimer's `init` pulses the peripheral's reset, which would make the timebase jump under any
+// other user, and `systimer.read` returns null when nothing has brought it up - neither is a
+// property a bus driver should impose on its caller.
+//
+// The CPU clock is whatever the bootloader left, measured at 90 MHz on this board and rated to
+// 400. Deadlines are computed at the 400 MHz *ceiling*, so on real silicon every timeout below is
+// between 1x and 4.4x longer than its nominal microseconds. That is the safe direction: a timeout
+// that fires early would turn a slow card into a spurious failure, and a timeout 4x long still
+// terminates.
+const assumed_cpu_hz_max: u32 = 400_000_000;
+
+inline fn cycleLow() u32 {
+ return asm volatile ("csrr %[r], 0xC00"
+ : [r] "=r" (-> u32),
+ );
+}
+
+/// A bounded wait. 32 bits of cycle counter wrap after 10.7 s at the assumed ceiling, which is an
+/// order of magnitude past the longest deadline here, and the wrapping subtraction is correct
+/// across the wrap anyway.
+const Deadline = struct {
+ start: u32,
+ budget: u32,
+
+ inline fn init(us: u32) Deadline {
+ return .{ .start = cycleLow(), .budget = us *% (assumed_cpu_hz_max / 1_000_000) };
+ }
+
+ inline fn expired(self: Deadline) bool {
+ return (cycleLow() -% self.start) >= self.budget;
+ }
+};
+
+/// `sd_host_private.h:62` SD_HOST_SDMMC_RESET_TIMEOUT_US.
+const reset_timeout_us: u32 = 5_000_000;
+/// `sd_host_private.h:61` SD_HOST_SDMMC_START_CMD_TIMEOUT_US - how long the CIU may take to accept
+/// a command word, which is a bus-side handshake and nothing to do with the card.
+const start_cmd_timeout_us: u32 = 1_000_000;
+/// How long to wait for the card's response after the command has been accepted. The controller
+/// has its own response timeout (TMOUT.response_timeout, 255 card clocks) and raises RTO, so this
+/// only has to cover the case where the controller itself never reports anything.
+const command_done_timeout_us: u32 = 200_000;
+/// Data phase. TMOUT.data_timeout is programmed to 100 ms of card clocks, matching
+/// `sd_host_sdmmc.c:531-533`; this outer bound is twice that.
+const data_done_timeout_us: u32 = 200_000;
+/// How long the card may hold DAT0 low before a new data command.
+const busy_timeout_us: u32 = 500_000;
+
+// ------------------------------------------------------------------------------------- geometry
+
+pub const Width = enum { one, four };
+
+/// Slot 1's pads on this board, and the GPIO-matrix signal each carries.
+///
+/// Slot 0 has a direct IO MUX function and slot 1 does not
+/// (`sdmmc_ll.h:88` SDMMC_LL_SLOT_SUPPORT_GPIO_MATRIX(1) is 1, and `sdmmc_periph.c:37-49` has
+/// -1 for every slot-1 IO MUX pin), so every slot-1 signal is routed through the matrix. The
+/// indices are `gpio_sig_map.h:8-18`, reached here through `regs` rather than written out: the
+/// same discipline `hal/gpio.zig` applies to SIG_GPIO_OUT_IDX, for the same reason.
+pub const Pins = struct {
+ clk: u8,
+ cmd: u8,
+ d0: u8,
+ d1: u8,
+ d2: u8,
+ d3: u8,
+};
+
+/// The ESP32-C6 coprocessor's wiring on this board. CLK 18, CMD 19, D0-D3 = 14/15/16/17.
+pub const c6_pins: Pins = .{ .clk = 18, .cmd = 19, .d0 = 14, .d1 = 15, .d2 = 16, .d3 = 17 };
+
+const sig = struct {
+ const cclk: u32 = @intCast(regs.SD_CARD_CCLK_2_PAD_OUT_IDX);
+ const ccmd: u32 = @intCast(regs.SD_CARD_CCMD_2_PAD_OUT_IDX);
+ const cdata0: u32 = @intCast(regs.SD_CARD_CDATA0_2_PAD_OUT_IDX);
+ const cdata1: u32 = @intCast(regs.SD_CARD_CDATA1_2_PAD_OUT_IDX);
+ const cdata2: u32 = @intCast(regs.SD_CARD_CDATA2_2_PAD_OUT_IDX);
+ const cdata3: u32 = @intCast(regs.SD_CARD_CDATA3_2_PAD_OUT_IDX);
+ const card_detect: u32 = @intCast(regs.SD_CARD_DETECT_N_2_PAD_IN_IDX);
+ const card_int: u32 = @intCast(regs.SD_CARD_INT_N_2_PAD_IN_IDX);
+
+ comptime {
+ // The `_2` in these names is slot 1: `sdmmc_periph.c:52-76` fills
+ // `sdmmc_slot_gpio_sig[1]` from exactly these macros. Slot 0's set is named `_1` and would
+ // route the wrong controller port to the C6's pads, silently.
+ std.debug.assert(cclk == 0 and ccmd == 1 and cdata0 == 2);
+ std.debug.assert(cdata1 == 3 and cdata2 == 4 and cdata3 == 5);
+
+ // The card interrupt is sensed on D1's *input* index, and `configurePins` hands `matrixIn`
+ // the *output* one - correct only because the P4's two signal tables agree on this signal.
+ // `gpio_sig_map.h:13-14` gives cdata1 the number 3 in both directions, and ESP-IDF relies
+ // on the same coincidence: `configure_pin_gpio_matrix` (`sd_host_sdmmc.c:1091-1105`) passes
+ // one `gpio_matrix_sig` to both `esp_rom_gpio_connect_in_signal` and `..._out_signal`.
+ // Asserted rather than assumed, because a mismatch here would route data correctly and
+ // sense interrupts from the wrong pad - which is invisible until something waits.
+ std.debug.assert(cdata1 == @as(u32, @intCast(regs.SD_CARD_CDATA1_2_PAD_IN_IDX)));
+ }
+};
+
+// -------------------------------------------------------------------------------------- state
+
+const State = struct {
+ slot: u1 = 1,
+ width: Width = .four,
+ /// The frequency `cardInit` switches to once the card is addressed and in 4-bit mode.
+ target_khz: u32 = 40_000,
+ pins: Pins = c6_pins,
+ /// Relative card address from CMD3, needed as the argument of CMD7.
+ rca: u16 = 0,
+ initialised: bool = false,
+};
+
+var state: State = .{};
+
+/// The card's relative address, as returned by CMD3. Zero until `cardInit` has run.
+pub fn rca() u16 {
+ return state.rca;
+}
+
+inline fn slotBit() u32 {
+ return @as(u32, 1) << state.slot;
+}
+
+// ------------------------------------------------------------------------------ command words
+//
+// One word, one function, no hardware. This is the part of the driver most worth testing on the
+// host, and the part the oracle can compare against ESP-IDF's own bitfield struct without going
+// anywhere near the card.
+
+/// Compose one field's contribution to a register word. `mmio.Reg.write` does this against a
+/// register; here the destination is a value, because the command word is built, checked and only
+/// then stored.
+inline fn bits(comptime f: Field, v: u32) u32 {
+ return (v & f.unshiftedMask()) << f.shift;
+}
+
+const cmd_index = Field.of(regs.SDHOST_CMD_INDEX_S, regs.SDHOST_CMD_INDEX_V);
+const response_expect = Field.of(regs.SDHOST_RESPONSE_EXPECT_S, regs.SDHOST_RESPONSE_EXPECT_V);
+const response_length = Field.of(regs.SDHOST_RESPONSE_LENGTH_S, regs.SDHOST_RESPONSE_LENGTH_V);
+const check_response_crc = Field.of(regs.SDHOST_CHECK_RESPONSE_CRC_S, regs.SDHOST_CHECK_RESPONSE_CRC_V);
+const data_expected = Field.of(regs.SDHOST_DATA_EXPECTED_S, regs.SDHOST_DATA_EXPECTED_V);
+const read_write = Field.of(regs.SDHOST_READ_WRITE_S, regs.SDHOST_READ_WRITE_V);
+const transfer_mode = Field.of(regs.SDHOST_TRANSFER_MODE_S, regs.SDHOST_TRANSFER_MODE_V);
+const send_auto_stop = Field.of(regs.SDHOST_SEND_AUTO_STOP_S, regs.SDHOST_SEND_AUTO_STOP_V);
+const wait_prvdata_complete = Field.of(regs.SDHOST_WAIT_PRVDATA_COMPLETE_S, regs.SDHOST_WAIT_PRVDATA_COMPLETE_V);
+const stop_abort_cmd = Field.of(regs.SDHOST_STOP_ABORT_CMD_S, regs.SDHOST_STOP_ABORT_CMD_V);
+const send_initialization = Field.of(regs.SDHOST_SEND_INITIALIZATION_S, regs.SDHOST_SEND_INITIALIZATION_V);
+const card_number = Field.of(regs.SDHOST_CARD_NUMBER_S, regs.SDHOST_CARD_NUMBER_V);
+const update_clock_registers_only = Field.of(regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_S, regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_V);
+/// `sdmmc_reg.h:486-494` spells this `USE_HOLE_REG`; `sdmmc_struct.h:473` spells it
+/// `use_hold_reg`, which is what it is - the hold register that synchronises CMD and DATA to
+/// cclk_out. Same bit 29, and ESP-IDF sets it on every command (`sd_host_sdmmc.c:859-860`).
+const use_hold_reg = Field.of(regs.SDHOST_USE_HOLE_REG_S, regs.SDHOST_USE_HOLE_REG_V);
+const start_cmd = Field.of(regs.SDHOST_START_CMD_S, regs.SDHOST_START_CMD_V);
+
+pub const Response = enum { none, short, long };
+pub const Direction = enum { read, write };
+
+/// Everything that distinguishes one command from another, in the terms the register uses.
+pub const Command = struct {
+ index: u6,
+ response: Response = .none,
+ /// Whether the controller checks the response's CRC7. Off for R3 and R4, which do not carry a
+ /// valid one - `sd_protocol_types.h:140-141` define both without SCF_RSP_CRC, and
+ /// `make_hw_cmd` (`sd_trans_sdmmc.c:214-216`) keys `check_response_crc` off exactly that flag.
+ check_crc: bool = false,
+ data: ?Direction = null,
+ /// 80 clocks of 1 before the command. Required once after power-on, and set only for CMD0,
+ /// which is where ESP-IDF sets it (`sd_trans_sdmmc.c:197-206`).
+ send_init: bool = false,
+ /// Wait for a previous data transfer to finish before sending. Set on everything except CMD0,
+ /// CMD12 and CMD11, again following `make_hw_cmd`.
+ wait_prvdata: bool = true,
+ auto_stop: bool = false,
+ stop_abort: bool = false,
+ /// Not a command at all: push CLKDIV/CLKSRC/CLKENA into the card clock domain.
+ update_clock: bool = false,
+ slot: u1 = 0,
+};
+
+/// The 32-bit word that, written to SDHOST_CMD_REG, issues `c`.
+///
+/// This is `make_hw_cmd` (`sd_trans_sdmmc.c:190-229`) plus the three fields
+/// `sd_host_slot_start_command` adds afterwards - `use_hold_reg`, `card_num` and `start_command`
+/// (`sd_host_sdmmc.c:859-881`) - because those three are not optional and splitting them across
+/// two functions is how one of them gets forgotten.
+pub fn commandWord(c: Command) u32 {
+ var w: u32 = 0;
+ w |= bits(cmd_index, c.index);
+ if (c.response != .none) w |= bits(response_expect, 1);
+ if (c.response == .long) w |= bits(response_length, 1);
+ if (c.check_crc) w |= bits(check_response_crc, 1);
+ if (c.data) |dir| {
+ w |= bits(data_expected, 1);
+ if (dir == .write) w |= bits(read_write, 1);
+ }
+ if (c.auto_stop) w |= bits(send_auto_stop, 1);
+ if (c.wait_prvdata) w |= bits(wait_prvdata_complete, 1);
+ if (c.stop_abort) w |= bits(stop_abort_cmd, 1);
+ if (c.send_init) w |= bits(send_initialization, 1);
+ if (c.update_clock) w |= bits(update_clock_registers_only, 1);
+ w |= bits(card_number, c.slot);
+ // Block transfers only; `transfer_mode` selects stream mode, which no SDIO command uses.
+ w |= bits(transfer_mode, 0);
+ w |= bits(use_hold_reg, 1);
+ w |= bits(start_cmd, 1);
+ return w;
+}
+
+// ------------------------------------------------------------------------------ SDIO protocol
+//
+// Command indices and argument layouts, from `sd_protocol_defs.h`. Written out as constants rather
+// than reached through `regs` because they are the SD specification, not this chip: the register
+// headers know nothing about them.
+
+/// `sd_protocol_defs.h:35`, `:40`, `:61`, `:78-80`.
+const cmd_go_idle_state: u6 = 0;
+const cmd_send_relative_addr: u6 = 3;
+const cmd_io_send_op_cond: u6 = 5;
+const cmd_select_card: u6 = 7;
+const cmd_io_rw_direct: u6 = 52;
+const cmd_io_rw_extended: u6 = 53;
+
+/// CMD52's argument: `sd_protocol_defs.h:484-492`.
+pub fn cmd52Arg(write: bool, func: u3, addr: u17, raw_flag: bool, data: u8) u32 {
+ var a: u32 = 0;
+ if (write) a |= @as(u32, 1) << 31;
+ a |= @as(u32, func) << 28;
+ if (raw_flag) a |= @as(u32, 1) << 27;
+ a |= @as(u32, addr) << 9;
+ a |= data;
+ return a;
+}
+
+/// CMD53's argument: `sd_protocol_defs.h:496-506`.
+///
+/// `count` is blocks in block mode and bytes in byte mode, and it is 9 bits: 0 means 512 in byte
+/// mode ("See 5.3.1 SDIO simplified spec", `sdmmc_io.c:351-355`) and infinite in block mode, which
+/// this driver never asks for.
+pub fn cmd53Arg(write: bool, func: u3, addr: u17, block_mode: bool, incrementing: bool, count: u9) u32 {
+ var a: u32 = 0;
+ if (write) a |= @as(u32, 1) << 31;
+ a |= @as(u32, func) << 28;
+ if (block_mode) a |= @as(u32, 1) << 27;
+ if (incrementing) a |= @as(u32, 1) << 26;
+ a |= @as(u32, addr) << 9;
+ a |= count;
+ return a;
+}
+
+/// The block size this driver programmes into BLKSIZ and into the card's CCCR/FBR.
+/// `sdmmc_common.h:195` SDMMC_IO_BLOCK_SIZE, and ESP-Hosted writes the same 512 into FN0 and FN1
+/// (`port_esp_hosted_host_sdio.c:211-217`).
+pub const io_block_size: u32 = 512;
+
+/// CCCR register offsets, `sd_protocol_defs.h:509-530`.
+pub const cccr = struct {
+ pub const revision: u17 = 0x00;
+ pub const fn_enable: u17 = 0x02;
+ pub const fn_ready: u17 = 0x03;
+ pub const int_enable: u17 = 0x04;
+ pub const int_pending: u17 = 0x05;
+ pub const ctl: u17 = 0x06;
+ pub const bus_width: u17 = 0x07;
+ pub const card_cap: u17 = 0x08;
+ pub const cis_ptr: u17 = 0x09;
+ pub const blksize_l: u17 = 0x10;
+ pub const blksize_h: u17 = 0x11;
+
+ pub const ctl_reset: u8 = 1 << 3;
+ pub const bus_width_1: u8 = 0;
+ pub const bus_width_4: u8 = 2;
+ /// Low-speed card; and "4-bit low speed", which says a low-speed card supports 4 bits anyway.
+ pub const card_cap_lsc: u8 = 1 << 6;
+ pub const card_cap_4bls: u8 = 1 << 7;
+};
+
+/// `sd_protocol_defs.h:533` SD_IO_FBR_START - function n's register block starts here.
+const fbr_start: u17 = 0x100;
+
+/// R4's fields, `sd_protocol_defs.h:478-481`.
+const r4_mem_ready: u32 = 1 << 31;
+const r4_mem_present: u32 = 1 << 27;
+
+/// The voltage window the host offers in CMD5's second pass: bits 23:15, i.e. 2.8-3.6 V.
+/// `sd_protocol_defs.h:109` SD_OCR_VOL_MASK, which is the whole of what `get_host_ocr` returns -
+/// "For now tell that the host has 2.8-3.6V voltage range" (`sdmmc_common.h:174-180`).
+const host_ocr: u32 = 0x00ff_8000;
+
+// ------------------------------------------------------------------------------- command issue
+
+/// Write one command word and wait for the CIU to take it. No card traffic is implied: a clock
+/// update command goes through here too.
+///
+/// Both waits are the ones `sd_host_slot_start_command` performs (`sd_host_sdmmc.c:862-892`),
+/// bounded the same way. The first is not redundant with the second: writing any command register
+/// while `start_command` is still set is a hardware locked write error, and HLE is reported
+/// asynchronously in RINTSTS where it is easy to attribute to the wrong command.
+fn startCommand(word: u32, arg: u32) Error!void {
+ var d = Deadline.init(start_cmd_timeout_us);
+ while (cmd.get(start_cmd) != 0) {
+ if (d.expired()) return error.Busy;
+ }
+ cmdarg.writeRaw(arg);
+ cmd.writeRaw(word);
+ d = Deadline.init(start_cmd_timeout_us);
+ while (cmd.get(start_cmd) != 0) {
+ if (d.expired()) return error.Timeout;
+ }
+}
+
+/// Push CLKDIV, CLKSRC and CLKENA into the card clock domain.
+fn clockUpdate() Error!void {
+ try startCommand(commandWord(.{
+ .index = 0,
+ .update_clock = true,
+ .wait_prvdata = true,
+ .slot = state.slot,
+ }), 0);
+}
+
+/// Turn a RINTSTS snapshot into the failure it describes.
+///
+/// Order matters only in that the first match wins, and it is chosen so the most specific cause is
+/// reported: a CRC error and a timeout together is a CRC error, because the timeout is downstream
+/// of it.
+fn decodeErrors(sts: u32) Error!void {
+ if (sts & (Event.rcrc | Event.dcrc) != 0) return error.CrcError;
+ if (sts & (Event.rto | Event.drto | Event.hto) != 0) return error.Timeout;
+ if (sts & (Event.re | Event.hle | Event.ebe | Event.sbe | Event.frun) != 0) return error.ResponseError;
+}
+
+/// Wait for one or more RINTSTS bits, failing on any error bit or on the deadline.
+///
+/// RINTSTS is write-1-to-clear, so this reads with `raw()` and clears with `writeRaw(mask)` -
+/// never `modify`, which would clear every bit it read back and lose the events this function is
+/// not waiting for.
+fn waitEvents(want: u32, errors: u32, us: u32) Error!u32 {
+ const d = Deadline.init(us);
+ while (true) {
+ const sts = rintsts.raw();
+ if (sts & errors != 0) {
+ rintsts.writeRaw(sts & (want | errors));
+ try decodeErrors(sts & errors);
+ // Every bit any caller passes in `errors` is covered above; a new one arriving here
+ // is a bug in this file, and reporting it beats an `unreachable` on a board with no
+ // debugger.
+ return error.ResponseError;
+ }
+ if (sts & want == want) {
+ rintsts.writeRaw(want);
+ return sts;
+ }
+ if (d.expired()) return error.Timeout;
+ }
+}
+
+/// A command with no data phase: issue it, wait for command-done, return R1/R5's first word.
+fn sendCommand(c: Command, arg: u32) Error!u32 {
+ // Everything this command is about to overwrite. This slot's SDIO card interrupt is
+ // deliberately left alone - the C6 raises it asynchronously and clearing it here would drop a
+ // wakeup the layer above is waiting for - and `clearNonSlaveInterrupts` is exactly that set.
+ //
+ // It used to be `Event.default & ~Event.cd`, which is a *subset* of the event bits and left
+ // four of them latched for ever: txdr(4), rxdr(5), frun(11) and acd(14). Two consequences, one
+ // cosmetic and one not. Cosmetic: every RINTSTS a diagnostic prints carries a stale 0x10 from
+ // the first transfer onwards, which is noise in exactly the register that has to be read
+ // carefully. Not cosmetic: **frun is a member of `Event.data_errors`**, so one FIFO
+ // under/overrun - ever - would latch a bit that nothing clears and fail every subsequent
+ // `waitEvents(Event.dto, Event.data_errors, ...)` for the rest of the run. The data path works
+ // today only because frun has never fired.
+ clearNonSlaveInterrupts();
+ var cc = c;
+ cc.slot = state.slot;
+ try startCommand(commandWord(cc), arg);
+ _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us);
+ return resp0.raw();
+}
+
+/// R5's status byte, the one CMD52 and CMD53 return. `sd_protocol_defs.h:493` takes the data byte;
+/// the flags above it say whether the card accepted the command at all.
+const r5_com_crc_error: u32 = 1 << 15;
+const r5_illegal_command: u32 = 1 << 14;
+const r5_error: u32 = 1 << 11;
+const r5_function_number: u32 = 1 << 9;
+const r5_out_of_range: u32 = 1 << 8;
+const r5_bad: u32 = r5_com_crc_error | r5_illegal_command | r5_error | r5_function_number | r5_out_of_range;
+
+fn checkR5(r: u32) Error!u8 {
+ if (r & r5_com_crc_error != 0) return error.CrcError;
+ if (r & r5_bad != 0) return error.ResponseError;
+ return @truncate(r);
+}
+
+// ------------------------------------------------------------------------------- bring-up
+
+/// Controller, FIFO and DMA reset, then wait for all three to self-clear.
+///
+/// All three bits are self-clearing, and `sdmmc_ll.h:486`, `:510` and `:534` each say so with a
+/// different delay ("two AHB clock cycles", "after reset done"). ESP-IDF sets all three and polls
+/// all three together (`sd_host_sdmmc.c:917-950`), which is what makes one bounded wait correct
+/// for the set.
+pub fn resetController() Error!void {
+ ctrl.modify(.{ controller_reset.is(1), fifo_reset.is(1), dma_reset.is(1) });
+ const d = Deadline.init(reset_timeout_us);
+ while (true) {
+ const v = ctrl.raw();
+ if (v & (controller_reset.mask() | fifo_reset.mask() | dma_reset.mask()) == 0) return;
+ if (d.expired()) return error.Timeout;
+ }
+}
+
+/// The interrupt configuration `sd_host_sdmmc.c:120-124` establishes - clear everything, mask
+/// everything, then unmask the completion and error events and turn the global enable on - with
+/// one deliberate deviation: card detect stays masked *and* gets cleared. See `Event.armed` for
+/// why that bit is load-bearing on a board with no card-detect pin.
+///
+/// `int_enable` gates the controller's single line into the CLIC. It is on even though this driver
+/// polls, because RINTSTS is set regardless and the layer above may register a handler for the
+/// SDIO card interrupt; leaving it off would mean `setSlaveInterruptEnabled(true)` silently did
+/// nothing.
+pub fn configureInterrupts() void {
+ rintsts.writeRaw(0xffff_ffff);
+ intmask.writeRaw(0);
+ ctrl.modify(.{int_enable.is(0)});
+ intmask.writeRaw(Event.armed);
+ // Belt and braces: `armed` keeps the controller from reporting a latched cd, and this makes
+ // sure there is no latched cd to report if anything ever unmasks it again.
+ rintsts.writeRaw(Event.cd);
+ ctrl.modify(.{int_enable.is(1)});
+}
+
+/// `sdmmc_ll_init_dma`, `sdmmc_ll.h:796-804`: enable the DMA path, clear the bus-mode register,
+/// pulse the IDMAC's own software reset, and unmask its three completion interrupts.
+pub fn initDma() void {
+ ctrl.modify(.{dma_enable.is(1)});
+ bmod.writeRaw(0);
+ bmod.modify(.{bmod_swr.is(1)});
+ idinten.modify(.{ idinten_ni.is(1), idinten_ri.is(1), idinten_ti.is(1) });
+}
+
+/// Leave the controller's interrupt output silent, and both status registers clean.
+///
+/// `configureInterrupts` and `initDma` above are ESP-IDF's sequences, and ESP-IDF is
+/// interrupt-driven: its transfers wait on a queue its ISR fills, so it needs command-done, the
+/// error bits and the IDMAC's completions in the masks. **This driver polls**, so every one of
+/// those is noise on a line whose only handler understands one cause. Worse than noise: two of
+/// them hold the line asserted forever.
+///
+/// * **INTMASK** gates RINTSTS into MINTSTS. Zero here costs nothing - `waitEvents` reads
+/// RINTSTS, and "Bits are logged regardless of interrupt mask status"
+/// (`sdmmc_struct.h:589-591`).
+/// * **IDINTEN** gates the IDMAC's own events, and it does *not* go through INTMASK. `initDma`
+/// enables NI/RI/TI because `sdmmc_ll_init_dma` does, and IDF can afford that because its ISR
+/// clears IDSTS on every interrupt (`sd_host_sdmmc.c:801-802`). `dataTransfer` clears IDSTS
+/// *before* a transfer and nothing clears it after, so RI and its sticky summary NIS stay set
+/// from the first CMD53 onwards - a permanently asserted interrupt line that no INTMASK write
+/// can lower.
+///
+/// `CTRL.int_enable` stays on: with both masks at zero the line cannot assert anyway, and leaving
+/// the global enable alone keeps `armSlaveInterrupt` down to the stores that matter.
+pub fn muteInterrupts() void {
+ intmask.writeRaw(0);
+ idinten.writeRaw(0);
+ rintsts.writeRaw(0xffff_ffff);
+ idsts.writeRaw(idsts_event_mask);
+}
+
+/// FIFO watermarks and DMA burst size.
+///
+/// ESP-IDF never writes this register on any target - there is no `sdmmc_ll` function for it and
+/// no assignment anywhere in `components/` - so the value in the measured working dump,
+/// `fifoth=0x01FF0000`, is the hardware's reset state: rx watermark 511, tx watermark 0, burst
+/// size code 0 (one transfer). This function writes that value explicitly rather than inheriting
+/// it, because a controller reset is not the only thing that can have touched the register and
+/// "the same as reset" is a claim worth making in code.
+///
+/// It is also a performance knob left deliberately untouched: DesignWare recommends half the FIFO
+/// depth for both watermarks and a burst size matching the AXI port, and tx watermark 0 means a
+/// DMA request only when the FIFO is completely empty. Turning that knob without a board to
+/// measure on would be guessing, and the guess would be against a configuration known to work at
+/// 40 MHz.
+pub fn setFifoThreshold(rx: u32, tx: u32, msize: u32) void {
+ fifoth.write(.{ rx_wmark.is(rx), tx_wmark.is(tx), dma_msize.is(msize) });
+}
+
+/// The reset-value watermarks, which are the ones the working dump shows.
+pub const default_rx_watermark: u32 = 511;
+pub const default_tx_watermark: u32 = 0;
+pub const default_dma_msize: u32 = 0;
+
+/// Bus width, host side. The card side is a CCCR write and is done in `cardInit`; the two must
+/// change in that order, or the next command goes out on a bus the card is not listening to.
+pub fn setBusWidth(w: Width) void {
+ const m = slotBit();
+ const c8 = ctype.get(card_width8) & ~m;
+ const c4 = switch (w) {
+ .one => ctype.get(card_width4) & ~m,
+ .four => ctype.get(card_width4) | m,
+ };
+ ctype.modify(.{ card_width4.is(c4), card_width8.is(c8) });
+}
+
+pub fn setBlockSize(bytes: u32) void {
+ blksiz.modify(.{block_size.is(bytes)});
+}
+
+/// The two-stage divider, resolved. Stage one is `host_div` in HP_SYS_CLKRST, stage two is the
+/// controller's own CLKDIV, and the card clock is `160 MHz / host_div / (2 * card_div)` with
+/// `card_div == 0` meaning bypass.
+///
+/// The table is `sd_host_slot_get_clk_dividers` (`sd_host_sdmmc.c:998-1062`), restricted to the
+/// PLL160M source: this board's C6 is a 3.3 V SDIO device, so the 200 MHz SDIO PLL and the UHS-I
+/// speeds it exists for are out of reach and out of scope.
+pub const Dividers = struct { host: u32, card: u32 };
+
+pub fn dividersFor(khz: u32) Dividers {
+ const src_hz: u32 = 160_000_000;
+ if (khz >= 40_000) return .{ .host = 4, .card = 0 }; // 160/4 = 40 MHz
+ if (khz == 20_000) return .{ .host = 8, .card = 0 }; // 160/8 = 20 MHz
+ if (khz == 400) return .{ .host = 10, .card = 20 }; // 160/10/(20*2) = 400 kHz
+ var host = src_hz / (khz * 1000);
+ var card: u32 = 0;
+ if (host > 15) {
+ host = 2;
+ card = (src_hz / 2) / (2 * khz * 1000);
+ if (((src_hz / 2) % (2 * khz * 1000)) > 0) card += 1;
+ } else if (src_hz % (khz * 1000) > 0) {
+ host += 1;
+ }
+ return .{ .host = host, .card = card };
+}
+
+/// Stage one: the clock generator in HP_SYS_CLKRST. `sdmmc_ll_set_clock_div`,
+/// `sdmmc_ll.h:244-258`.
+///
+/// The `edge_cfg_update` bit is write-to-trigger and must be pulsed - set then cleared - after the
+/// three edge fields, or the new division is programmed and never latched.
+pub fn setHostClockDiv(div: u32) void {
+ if (div > 1) {
+ peri_clk_ctrl02.modify(.{
+ sdio_ls_clk_edge_h.is(div / 2 - 1),
+ sdio_ls_clk_edge_n.is(div - 1),
+ sdio_ls_clk_edge_l.is(div - 1),
+ });
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)});
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)});
+ } else {
+ peri_clk_ctrl01.modify(.{sdio_hs_mode.is(1)});
+ peri_clk_ctrl02.modify(.{
+ sdio_ls_clk_edge_h.is(0),
+ sdio_ls_clk_edge_n.is(0),
+ sdio_ls_clk_edge_l.is(0),
+ });
+ }
+}
+
+/// PLL160M, the only source this driver uses. `sdmmc_ll_select_clk_source`, `sdmmc_ll.h:212-229`:
+/// source value 0 is PLL160M and 1 is the 200 MHz SDIO PLL.
+pub fn selectPll160m() void {
+ peri_clk_ctrl01.modify(.{ sdio_ls_clk_src_sel.is(0), sdio_ls_clk_en.is(1) });
+}
+
+/// The driving, sampling and self clocks the pad logic runs on. `sdmmc_ll_init_phase_delay`,
+/// `sdmmc_ll.h:303-315`. Without this the three gates stay off and the bus does not move, which is
+/// the kind of failure that looks like a wiring fault.
+pub fn initPhaseDelay() void {
+ peri_clk_ctrl02.modify(.{
+ sdio_ls_drv_clk_en.is(1),
+ sdio_ls_sam_clk_en.is(1),
+ sdio_ls_slf_clk_en.is(1),
+ sdio_ls_drv_clk_edge_sel.is(1),
+ sdio_ls_sam_clk_edge_sel.is(0),
+ sdio_ls_slf_clk_edge_sel.is(0),
+ });
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)});
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)});
+}
+
+/// Stage one, whole: divider, source, phase clocks, and the settle the hardware needs afterwards.
+/// `sd_host_set_clk_div`, `sd_host_sdmmc.c:974-990`, including its closing
+/// `esp_rom_delay_us(10)` - "Wait for the clock to propagate".
+///
+/// This has to happen before the controller reset, not after. `controller_reset` is documented to
+/// self-clear "after two AHB and two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`), so with
+/// no card clock reaching the block the bit never clears and the reset wait runs to its full
+/// timeout. ESP-IDF's order says the same thing without saying it: `sd_host_set_clk_div` at
+/// `sd_host_sdmmc.c:109`, `sd_host_reset` at `:112`.
+pub fn setHostClock(div: u32) void {
+ setHostClockDiv(div);
+ selectPll160m();
+ initPhaseDelay();
+ spinMicros(10);
+}
+
+/// Stage two: the controller's per-slot divider and the divider-to-slot mux.
+/// `sdmmc_ll_set_card_clock_div`, `sdmmc_ll.h:431-442`. Slot 1 uses divider 1, slot 0 uses divider
+/// 0 - so the mux value equals the slot number, which is why one line covers both.
+pub fn setCardClockDiv(div: u32) void {
+ if (state.slot == 0) {
+ clksrc.modify(.{clksrc_card0.is(0)});
+ clkdiv.modify(.{clk_divider0.is(div)});
+ } else {
+ clksrc.modify(.{clksrc_card1.is(1)});
+ clkdiv.modify(.{clk_divider1.is(div)});
+ }
+}
+
+/// The card clock's on/off switch, one bit per slot. Takes effect only after a clock update
+/// command. `sdmmc_ll_enable_card_clock`, `sdmmc_ll.h:415-422`.
+pub fn setCardClockEnabled(on: bool) void {
+ const cur = clkena.get(cclk_enable);
+ clkena.modify(.{cclk_enable.is(if (on) cur | slotBit() else cur & ~slotBit())});
+}
+
+/// Stop the card clock while the card is idle. `sdmmc_ll_enable_card_clock_low_power`,
+/// `sdmmc_ll.h:474-481`. **Off** for SDIO: the card raises its interrupt on D1 and cannot do so
+/// with the clock stopped, which is why ESP-IDF clears the same bit for any slot with
+/// `cclk_always_on` (`sd_host_sdmmc.c:272-285`) and why the measured working dump reads
+/// `clkena=0x00000002` rather than `0x00020002`.
+pub fn setCardClockLowPower(on: bool) void {
+ const cur = clkena.get(lp_enable);
+ clkena.modify(.{lp_enable.is(if (on) cur | slotBit() else cur & ~slotBit())});
+}
+
+/// Bytes in the next data transfer. `sdmmc_ll_set_data_transfer_len`, `sdmmc_ll.h:651-654`.
+pub fn setDataTransferLen(len: u32) void {
+ bytcnt.modify(.{byte_count.is(len)});
+}
+
+/// Data-read and response timeouts, both in card output clocks.
+/// `sdmmc_ll_set_data_timeout` / `sdmmc_ll_set_response_timeout`, `sdmmc_ll.h:564-582`.
+pub fn setTimeouts(data_cycles: u32, response_cycles: u32) void {
+ tmout.write(.{
+ data_timeout.is(if (data_cycles > 0xff_ffff) 0xff_ffff else data_cycles),
+ response_timeout.is(response_cycles),
+ });
+}
+
+/// Turn the internal DMA path on or off: both CTRL bits and both BMOD bits, together.
+/// `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`.
+pub fn setDmaEnabled(on: bool) void {
+ const v: u32 = @intFromBool(on);
+ ctrl.modify(.{ dma_enable.is(v), use_internal_dma.is(v) });
+ bmod.modify(.{ bmod_de.is(v), bmod_fb.is(v) });
+}
+
+/// Where the IDMAC fetches its first descriptor. `sdmmc_ll_set_desc_addr`, `sdmmc_ll.h:673-676`.
+/// The address is the *cached* one; see the "Cache" section above for why that is right.
+pub fn setDescriptorAddr(a: u32) void {
+ dbaddr.writeRaw(a);
+}
+
+/// Change the card clock, safely: stop it, reprogramme both stages, start it again, with a clock
+/// update command after each step. `sd_host_slot_set_card_clk`, `sd_host_sdmmc.c:487-537`.
+///
+/// Low-power mode is left **off**, which is the one place this deviates from a plain SD host and
+/// matches the measured dump (`clkena=0x00000002`: clock enabled for slot 1, `lp_enable` clear).
+/// `clkena.lp_enable` stops cclk while the card is idle; an SDIO card signals its interrupt on D1
+/// and needs the clock running to do it, which is why ESP-IDF turns the same bit off for any slot
+/// with `cclk_always_on` (`sd_host_sdmmc.c:272-285`).
+pub fn setBusClock(khz: u32) Error!void {
+ const d = dividersFor(khz);
+
+ setCardClockEnabled(false);
+ try clockUpdate();
+
+ setCardClockDiv(d.card);
+ setHostClock(d.host);
+ try clockUpdate();
+
+ setCardClockEnabled(true);
+ setCardClockLowPower(false);
+ try clockUpdate();
+
+ // 100 ms of card clocks for data, and the maximum 255 card clocks for a response - "always set
+ // response timeout to highest value, it's small enough anyway" (`sd_host_sdmmc.c:534-535`).
+ setTimeouts(100 * khz, 255);
+}
+
+/// Route slot 1's six signals to the C6's pads.
+///
+/// Pull-ups: **the board provides them externally and this enables the internal ones anyway**, on
+/// all six pads, because that is what the working configuration does. It is not obvious from
+/// ESP-Hosted's side - it leaves `SDMMC_SLOT_FLAG_INTERNAL_PULLUP` clear
+/// (`SDMMC_SLOT_CONFIG_DEFAULT`, `sdmmc_default_configs.h:98`: `.flags = 0`) - but every pad still
+/// gets one, because `configure_pin_gpio_matrix` opens with `gpio_reset_pin`
+/// (`sd_host_sdmmc.c:1096`) and that function enables the pull-up unconditionally: "for powersave
+/// reasons, the GPIO should not be floating, select pullup" (`gpio.c:469-472`). The 40 MHz link
+/// that produced the register dump therefore had both the module's external pull-ups and these.
+/// Matching a measured configuration beats reasoning about which resistor is redundant.
+///
+/// D1 has a second job: it is the SDIO interrupt line, and the controller derives that interrupt
+/// from the same routed data signal - `sd_host_slot_sdmmc_io_int_enable` (`sd_host_sdmmc.c:381-388`)
+/// is *only* `configure_pin(d1, sdmmc_slot_gpio_sig[slot].d1, GPIO_MODE_INPUT_OUTPUT)`, the same
+/// two matrix writes and the same `fun_ie` the loop below already does, and it touches no
+/// controller register at all. Both halves are load-bearing and neither is visible in a working
+/// data path: with `fun_ie` clear, or with the *input* side of the matrix left pointing elsewhere,
+/// D1 still drives and every transfer still completes while the controller samples a constant and
+/// never latches a card interrupt. A link that carries traffic and never reports an event is
+/// exactly what that failure looks like, which is why D1 is routed both ways even in 1-bit mode
+/// and why `interruptDiagnostics` prints both bits.
+///
+/// D3 is *not* routed to the controller yet. It is driven high as a plain GPIO output until the
+/// bus is switched to 4 bits, which is how a host tells an SDIO card to use SD mode rather than
+/// SPI mode; `sd_host_sdmmc.c:1282-1294` does the same and `cardInit` reconnects it at
+/// `sd_host_sdmmc.c:575-583`'s point in the sequence.
+pub fn configurePins(pins: Pins) void {
+ // CLK is output-only.
+ gpio.matrixOut(pins.clk, sig.cclk);
+ gpio.setInputEnable(pins.clk, false);
+ gpio.setPull(pins.clk, .up);
+
+ const bidir = [_]struct { pin: u8, signal: u32 }{
+ .{ .pin = pins.cmd, .signal = sig.ccmd },
+ .{ .pin = pins.d0, .signal = sig.cdata0 },
+ .{ .pin = pins.d1, .signal = sig.cdata1 },
+ .{ .pin = pins.d2, .signal = sig.cdata2 },
+ };
+ for (bidir) |b| {
+ gpio.matrixOut(b.pin, b.signal);
+ gpio.matrixIn(b.pin, b.signal);
+ gpio.setInputEnable(b.pin, true);
+ gpio.setPull(b.pin, .up);
+ }
+
+ // D3 high, as a GPIO, until the bus width changes.
+ gpio.configureOutput(pins.d3, .{ .readback = true });
+ gpio.setPull(pins.d3, .up);
+ gpio.setHigh(pins.d3);
+
+ // Card detect and the card's own interrupt-request pin are not wired to anything on this
+ // board, so both are tied off in the matrix exactly as ESP-IDF ties them when no pin is
+ // configured: card-detect to a constant 0 ("card present", `sd_host_sdmmc.c:1315-1319`) and
+ // card-int-n to a constant 1, i.e. inactive (`:1304-1306`). Leaving them unrouted is not the
+ // same thing: GPIO_FUNCn_IN_SEL_CFG resets with `sig_in_sel` clear, which bypasses the matrix
+ // and takes the signal from whatever direct pad function exists - and slot 1 has none.
+ // Write protect is left alone; nothing in this driver reads WRTPRT.
+ gpio.matrixIn(gpio.matrix_const_zero, sig.card_detect);
+ gpio.matrixIn(gpio.matrix_const_one, sig.card_int);
+}
+
+/// Reconnect D3 to the controller, once the card is in 4-bit mode.
+fn attachD3() void {
+ gpio.matrixOut(state.pins.d3, sig.cdata3);
+ gpio.matrixIn(state.pins.d3, sig.cdata3);
+ gpio.setInputEnable(state.pins.d3, true);
+ gpio.setPull(state.pins.d3, .up);
+}
+
+/// Everything from the clock gate to a controller that will accept a command, with the bus at the
+/// 400 kHz probing frequency and 1 bit wide - which is where an SDIO card has to be met.
+///
+/// `cardInit` is what raises it to `khz` and to `width`, after the card has been addressed.
+pub fn init(opts: struct {
+ slot: u1 = 1,
+ width: Width = .four,
+ khz: u32 = 40_000,
+ pins: Pins = c6_pins,
+}) Error!void {
+ state = .{
+ .slot = opts.slot,
+ .width = opts.width,
+ .target_khz = opts.khz,
+ .pins = opts.pins,
+ };
+
+ // The C6 hangs off slot 1 and slot 0's pads are the P4's own flash on most boards; refusing
+ // here is cheaper than debugging a bricked boot.
+ if (opts.slot != 1) return error.NotSupported;
+
+ // 1. Bus clock and reset. Unlike most of this chip, SDMMC's bus clock is gated *off* at
+ // power-on (HP_SYS_CLKRST SOC_CLK_CTRL1 REG_SDMMC_SYS_CLK_EN, default 0), so this is a
+ // prerequisite and not a formality - without it the register block reads stale nonsense.
+ // Its reset bit is not in HP_SYS_CLKRST at all but in LP_AON_CLKRST; see hal/clkrst.zig.
+ clkrst.init(.sdmmc);
+
+ // 2. The host clock generator, *before* the controller reset and not after. `sd_host_reset`
+ // polls three self-clearing bits, and `controller_reset` clears only "after two AHB and
+ // two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`) - with no card clock reaching the
+ // block that poll runs to its full timeout. ESP-IDF's controller init has the same order:
+ // `sd_host_set_clk_div(ctlr, SDMMC_CLK_SRC_DEFAULT, 2)` at `sd_host_sdmmc.c:109`, then
+ // `sd_host_reset` at `:112`. Divider 2 is IDF's provisional value, replaced at step 6.
+ setHostClock(2);
+
+ // 3. Controller, FIFO and DMA out of reset.
+ try resetController();
+
+ // 4. Interrupts and DMA, before any command can produce one. The first two reproduce ESP-IDF
+ // for the differential; `muteInterrupts` then takes back everything this driver polls for
+ // instead of being interrupted by, leaving the line into the CLIC silent until a waiter
+ // arms it.
+ configureInterrupts();
+ initDma();
+ muteInterrupts();
+
+ // 5. Pads. After the clock gate so the controller's outputs are real, before the card clock so
+ // the first cycle the C6 sees is a clean one.
+ configurePins(state.pins);
+
+ // 6. Bus clock at probing speed, 1 bit wide. An SDIO card has to be met at 400 kHz in 1-bit
+ // mode; `cardInit` raises both once the card has been addressed.
+ try setBusClock(400);
+ setBusWidth(.one);
+
+ // 7. Transfer geometry.
+ setBlockSize(io_block_size);
+ setFifoThreshold(default_rx_watermark, default_tx_watermark, default_dma_msize);
+ setDescriptorAddr(busAddr(&dma.desc));
+
+ // 8. The one cache operation in this driver's life. See the "Cache" section above.
+ syncDmaRegionOnce();
+
+ state.initialised = true;
+
+ // The one claim in this sequence with no differential case behind it, said out loud once, at
+ // the moment it is true. `configureInterrupts` and `initDma` are compared against ESP-IDF on
+ // the die; `muteInterrupts` cannot be - it is a deliberate deviation from IDF's ISR-driven
+ // design, and a reference implementation of our own decision would prove nothing. This line is
+ // the substitute, and it is worth a print because both zeros are load-bearing: a non-zero
+ // idinten here is an interrupt line that no INTMASK write can ever lower.
+ note("MARK SDMMC_INIT intmask=0x%08x idinten=0x%08x expect 0x00000000 and 0x00000000\r\n", .{
+ intmask.raw(), idinten.raw(),
+ });
+}
+
+// ------------------------------------------------------------------------------ card bring-up
+
+/// CMD0, CMD5, CMD3, CMD7, then the CCCR writes that make function 1 usable: the sequence that
+/// takes the C6 from "powered" to "answers CMD52".
+///
+/// The command half follows `sdmmc_card_init` (`sdmmc_init.c:78-133`) restricted to the SDIO path:
+/// `sdmmc_io_reset`, CMD0, `sdmmc_init_io` (CMD5 twice), `sdmmc_init_rca` (CMD3),
+/// `sdmmc_init_select_card` (CMD7). The CCCR half is ESP-Hosted's `hosted_sdio_card_fn_init`
+/// (`port_esp_hosted_host_sdio.c:143-220`) - enable function 1, wait for it to report ready,
+/// unmask its interrupt, switch to 4 bits, set both block sizes to 512 - because that is what this
+/// particular device needs and IDF's generic SDIO init does not do.
+///
+/// The CMD52 that resets the card is allowed to fail. A device that is already out of reset
+/// answers it; one that is not may time out, and `sdmmc_io_reset` (`sdmmc_io.c:66-83`) accepts
+/// exactly that.
+pub fn cardInit() Error!void {
+ if (!state.initialised) return error.NotSupported;
+
+ // CCCR CTL bit 3: I/O reset. Best-effort, as above.
+ cmd52Write(0, cccr.ctl, cccr.ctl_reset) catch {};
+
+ // CMD0 with the 80-clock init sequence and no response.
+ _ = try sendCommand(.{
+ .index = cmd_go_idle_state,
+ .response = .none,
+ .send_init = true,
+ .wait_prvdata = false,
+ }, 0);
+ // SDMMC_GO_IDLE_DELAY_MS (`sdmmc_common.h:34`), which `sdmmc_send_cmd_go_idle_state` waits
+ // out before returning (`sdmmc_cmd.c:114-116`). CMD0 has no response, so there is nothing to
+ // wait *for*: this is the card's own settling time and skipping it makes the next command a
+ // coin toss.
+ spinMicros(20_000);
+
+ // CMD5 with a zero argument asks "are you an IO card, and what voltages do you take"; R4 has
+ // no CRC, hence `check_crc = false` (`sd_protocol_types.h:141`).
+ const probe = try sendCommand(.{
+ .index = cmd_io_send_op_cond,
+ .response = .short,
+ .check_crc = false,
+ }, 0);
+ const functions = (probe >> 28) & 0x7;
+ if (functions == 0) return error.NotSupported; // answered CMD5, but has no IO function
+
+ // CMD5 again with the voltage window, until the card reports ready. 100 attempts is
+ // `sdmmc_io.c:240`; the 10 ms between them is SDMMC_IO_SEND_OP_COND_DELAY_MS
+ // (`sdmmc_common.h:35`), spent here as a bounded spin rather than a scheduler delay.
+ const ocr = host_ocr & probe;
+ var ready = false;
+ var tries: u32 = 0;
+ while (tries < 100) : (tries += 1) {
+ const r = try sendCommand(.{
+ .index = cmd_io_send_op_cond,
+ .response = .short,
+ .check_crc = false,
+ }, ocr);
+ if (r & r4_mem_ready != 0) {
+ ready = true;
+ break;
+ }
+ spinMicros(10_000);
+ }
+ if (!ready) return error.Timeout;
+
+ // CMD3: the card picks its own relative address and returns it in R6[31:16].
+ const r6 = try sendCommand(.{
+ .index = cmd_send_relative_addr,
+ .response = .short,
+ .check_crc = true,
+ }, 0);
+ state.rca = @truncate(r6 >> 16);
+
+ // CMD7 with that address moves the card from stand-by to transfer state. Every CMD52 and
+ // CMD53 after this is addressed to it implicitly.
+ _ = try sendCommand(.{
+ .index = cmd_select_card,
+ .response = .short,
+ .check_crc = true,
+ }, @as(u32, state.rca) << 16);
+
+ // ---- CCCR: function 1 on.
+ const ioe = try cmd52Read(0, cccr.fn_enable);
+ try cmd52Write(0, cccr.fn_enable, ioe | 0x02);
+
+ // Wait for IOR bit 1. ESP-Hosted polls with a 10 ms gap and gives up after SDIO_INIT_MAX_RETRY
+ // (`port_esp_hosted_host_sdio.c:177-192`).
+ var fn_ready = false;
+ tries = 0;
+ while (tries < 100) : (tries += 1) {
+ if ((try cmd52Read(0, cccr.fn_ready)) & 0x02 != 0) {
+ fn_ready = true;
+ break;
+ }
+ spinMicros(10_000);
+ }
+ if (!fn_ready) return error.Timeout;
+
+ // Master interrupt enable plus function 1's, so the C6 can raise D1.
+ const ie = try cmd52Read(0, cccr.int_enable);
+ try cmd52Write(0, cccr.int_enable, ie | 0x01 | 0x02);
+
+ // ---- Bus width: card first, then host, then D3 joins the bus.
+ if (state.width == .four) {
+ const cap = try cmd52Read(0, cccr.card_cap);
+ // "Not a low-speed card" or "a low-speed card that supports 4 bits" - `sdmmc_io.c:182-183`.
+ if ((cap & cccr.card_cap_lsc) == 0 or (cap & cccr.card_cap_4bls) != 0) {
+ try cmd52Write(0, cccr.bus_width, cccr.bus_width_4);
+ setBusWidth(.four);
+ attachD3();
+ } else {
+ state.width = .one;
+ }
+ }
+
+ // ---- Block size 512 for function 0 and function 1, host side and card side.
+ try setCardBlockSize(0, io_block_size);
+ try setCardBlockSize(1, io_block_size);
+ setBlockSize(io_block_size);
+
+ // ---- Finally the target frequency, now that the card is addressed and the bus is wide.
+ try setBusClock(state.target_khz);
+}
+
+/// The 16-bit block size lives in two consecutive byte registers, low half first
+/// (`port_esp_hosted_host_sdio.c:123-141`). Function n's copy is at `0x100 * n + 0x10`.
+fn setCardBlockSize(func: u3, bytes: u16) Error!void {
+ const base: u17 = fbr_start * @as(u17, func);
+ try cmd52Write(0, base + cccr.blksize_l, @truncate(bytes));
+ try cmd52Write(0, base + cccr.blksize_h, @truncate(bytes >> 8));
+}
+
+/// A bounded busy-wait, for the two places the SDIO specification asks for a delay between
+/// retries. Same conservative frequency assumption as `Deadline`, in the same safe direction: on
+/// this 90 MHz die a 10 ms request takes about 44 ms.
+fn spinMicros(us: u32) void {
+ const d = Deadline.init(us);
+ while (!d.expired()) {}
+}
+
+// ------------------------------------------------------------------------------------- CMD52
+
+/// Read one byte from the card's register space.
+///
+/// This is the whole minimal milestone: after `init` and `cardInit`, `cmd52Read(0, 0x00)` reads
+/// CCCR offset 0 and the byte that comes back is the C6 answering.
+pub fn cmd52Read(func: u3, addr: u17) Error!u8 {
+ const r = try sendCommand(.{
+ .index = cmd_io_rw_direct,
+ .response = .short,
+ .check_crc = true,
+ }, cmd52Arg(false, func, addr, false, 0));
+ return checkR5(r);
+}
+
+/// Write one byte. The RAW flag is not set, matching `sdmmc_io_rw_direct` with SD_ARG_CMD52_WRITE
+/// alone (`sdmmc_io.c:187`); `sdmmc_io_write_byte` adds SD_ARG_CMD52_EXCHANGE when it wants the
+/// previous value back, which no caller here does.
+pub fn cmd52Write(func: u3, addr: u17, value: u8) Error!void {
+ const r = try sendCommand(.{
+ .index = cmd_io_rw_direct,
+ .response = .short,
+ .check_crc = true,
+ }, cmd52Arg(true, func, addr, false, value));
+ _ = try checkR5(r);
+}
+
+// ------------------------------------------------------------------------------------- CMD53
+
+/// How one CMD53 is split. Two rules decide it, and both come from ESP-IDF rather than from the
+/// SDIO specification, because both are properties of this controller:
+///
+/// * **Block mode when the length is a whole number of 512-byte blocks**, byte mode otherwise.
+/// In byte mode the count field is bytes and 0 encodes 512 ("See 5.3.1 SDIO simplified spec",
+/// `sdmmc_io.c:351-355`), so one byte-mode command reaches 512 bytes and no further.
+/// * **A byte-mode length of 4 or more must be a multiple of 4.** `sd_trans_sdmmc.c:526-532`
+/// rejects anything else outright, and `sdmmc_io_read_bytes` works around it by splitting:
+/// "host quirk: SDIO transfer with length not divisible by 4 bytes has to be split into two
+/// transfers: one with aligned length, the other one for the remaining 1-3 bytes"
+/// (`sdmmc_io.c:400-419`). So 6 bytes is two commands, 4 then 2, and 3 bytes is one.
+///
+/// A caller that wants the split to be explicit - ESP-Hosted's block path does, because its
+/// addresses increment across the split - can hand over one whole-block chunk at a time and get
+/// exactly one block-mode command per call. A caller that does not can hand over any length.
+const Chunk = struct {
+ block_mode: bool,
+ /// Bytes in this command.
+ len: u32,
+ /// The CMD53 count field: blocks in block mode, bytes in byte mode with 0 meaning 512.
+ count: u9,
+};
+
+fn nextChunk(remaining: u32) Chunk {
+ if (remaining >= io_block_size and remaining % io_block_size == 0) {
+ const max_blocks = bounce_len / io_block_size;
+ var blocks = remaining / io_block_size;
+ if (blocks > max_blocks) blocks = max_blocks;
+ return .{
+ .block_mode = true,
+ .len = blocks * io_block_size,
+ .count = @intCast(blocks),
+ };
+ }
+ var len = remaining;
+ if (len > io_block_size) len = io_block_size;
+ // The 4-byte rule. Below 4 bytes the whole request goes in one command; at or above it, the
+ // aligned part goes first and the 1-3 byte tail becomes the next chunk.
+ if (len >= 4 and len % 4 != 0) len &= ~@as(u32, 3);
+ return .{
+ .block_mode = false,
+ .len = len,
+ .count = if (len == io_block_size) 0 else @intCast(len),
+ };
+}
+
+pub fn cmd53Read(func: u3, addr: u17, buf: []u8, incrementing: bool) Error!void {
+ var offset: u32 = 0;
+ var a: u32 = addr;
+ while (offset < buf.len) {
+ const c = nextChunk(@intCast(buf.len - offset));
+ const arg = cmd53Arg(false, func, @truncate(a), c.block_mode, incrementing, c.count);
+ try dataTransfer(.read, arg, c.len, if (c.block_mode) io_block_size else c.len);
+ const dst = buf[offset..][0..c.len];
+ const src = bufNc();
+ for (dst, 0..) |*b, i| b.* = src[i];
+ offset += c.len;
+ if (incrementing) a += c.len;
+ }
+}
+
+pub fn cmd53Write(func: u3, addr: u17, data: []const u8, incrementing: bool) Error!void {
+ var offset: u32 = 0;
+ var a: u32 = addr;
+ while (offset < data.len) {
+ const c = nextChunk(@intCast(data.len - offset));
+ const src = data[offset..][0..c.len];
+ const dst = bufNc();
+ for (src, 0..) |b, i| dst[i] = b;
+ // The IDMAC moves whole words, so a length that is not a multiple of 4 is rounded up
+ // (`sd_trans_sdmmc.c:127`). Zero the pad rather than send whatever the last transfer left.
+ var pad = c.len;
+ while (pad % 4 != 0) : (pad += 1) dst[pad] = 0;
+ const arg = cmd53Arg(true, func, @truncate(a), c.block_mode, incrementing, c.count);
+ try dataTransfer(.write, arg, c.len, if (c.block_mode) io_block_size else c.len);
+ offset += c.len;
+ if (incrementing) a += c.len;
+ }
+}
+
+/// One CMD53 with its data phase, through the IDMAC and the bounce buffer.
+///
+/// Order is ESP-IDF's (`sd_trans_sdmmc.c:524-568`): descriptor and transfer registers first, then
+/// the command word, then wait for command-done and data-transfer-over in that order. Preparing
+/// the DMA after starting the command would be a race against a card that answers immediately.
+fn dataTransfer(dir: Direction, arg: u32, len: u32, blk: u32) Error!void {
+ std.debug.assert(len <= bounce_len);
+
+ // The card must not still be holding DAT0 low from a previous write.
+ const busy = Deadline.init(busy_timeout_us);
+ while (status.get(data_busy) != 0) {
+ if (busy.expired()) return error.Busy;
+ }
+
+ // As in `sendCommand`: the whole event set except this slot's card interrupt. `frun` is in
+ // `Event.data_errors` and nothing else ever clears it.
+ clearNonSlaveInterrupts();
+ idsts.writeRaw(idsts_event_mask);
+
+ const padded = (len + 3) & ~@as(u32, 3);
+ const d = descNc();
+ d.buffer1 = busAddr(&dma.buf);
+ d.next = 0;
+ d.sizes = padded; // buffer1_size is [12:0]; buffer2 is unused
+ d.flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor |
+ Descriptor.last_descriptor | Descriptor.second_address_chained;
+
+ setDataTransferLen(len);
+ setBlockSize(blk);
+ setDescriptorAddr(busAddr(&dma.desc));
+
+ // `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`, then the poll demand that tells the IDMAC to
+ // re-read a descriptor it may have parked on.
+ setDmaEnabled(true);
+ pldmnd.writeRaw(1);
+
+ try startCommand(commandWord(.{
+ .index = cmd_io_rw_extended,
+ .response = .short,
+ .check_crc = true,
+ .data = dir,
+ .slot = state.slot,
+ }), arg);
+
+ _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us);
+ _ = try checkR5(resp0.raw());
+ _ = try waitEvents(Event.dto, Event.data_errors, data_done_timeout_us);
+}
+
+// -------------------------------------------------------------------- SDIO card interrupt (D1)
+
+/// Has the card asserted its interrupt line?
+///
+/// Non-blocking, no side effect, straight out of RINTSTS bit 16+slot (`sdmmc_reg.h:621-631`). It
+/// does **not** clear the bit; `clearSlaveInterrupt` does, deliberately, once a caller has decided
+/// to act on it.
+///
+/// Two different trigger behaviours meet at this bit and it is worth keeping them apart, because
+/// conflating them sends you tuning the wrong knob:
+///
+/// * **Card -> controller is an edge.** ESP-IDF: "SDIO interrupts are negedge sensitive ones:
+/// the status bit is only set when first interrupt triggered" (`sd_host_sdmmc.c:396-402`).
+/// That is why a waiter must check D1's level once before sleeping - an edge that arrived
+/// while it was awake is not re-delivered.
+/// * **Controller -> CLIC is a level.** RINTSTS is a sticky write-1-to-clear latch, so the
+/// controller's output line stays asserted until software clears the bit that raised it. The
+/// CLIC line therefore wants `.level`, and an edge trigger there would only hide a handler
+/// that fails to deassert rather than fix it.
+///
+/// This is the polling half. The interrupt half is a CLIC line and belongs to whoever owns the
+/// scheduler: route `interrupt_source` with `hal.intr`, and in the handler mask the bit
+/// (`setSlaveInterruptEnabled(false)`) before waking anybody. ESP-IDF does exactly that
+/// (`sd_host_sdmmc.c:826-830`) and explains why at `:396-402`: "SDIO interrupts are negedge
+/// sensitive ones: the status bit is only set when first interrupt triggered", so a handler that
+/// leaves the bit unmasked and unhandled re-enters forever, and a waiter that sleeps without first
+/// checking D1's level loses an edge that arrived while it was awake.
+pub fn slaveInterruptPending() bool {
+ return rintsts.raw() & (Event.io_slot0 << state.slot) != 0;
+}
+
+/// The raw masked-interrupt status word. Diagnostics only: a hang waiting on the card interrupt is
+/// otherwise indistinguishable from a card that never asserted, and this is the register that tells
+/// them apart.
+pub fn interruptStatusRaw() u32 {
+ return rintsts.raw();
+}
+
+pub fn clearSlaveInterrupt() void {
+ rintsts.writeRaw(Event.io_slot0 << state.slot);
+}
+
+/// INTMASK as written. The other half of "why is this line asserted": the controller's output is
+/// RINTSTS AND INTMASK, and a diagnostic that prints only RINTSTS shows half the conjunction.
+///
+/// **Not evidence about an arm.** INTMASK is an ordinary read/write register, so this returns
+/// whatever the last store left - and on the interrupt path the last store is usually the *disarm*.
+/// A caller that wants to know whether unmasking took effect must read the register back inside the
+/// same masked region as the store; that is `armSlaveInterrupt`, and it exists because this
+/// function was read as if it answered that question and it never could.
+pub fn interruptMaskRaw() u32 {
+ return intmask.raw();
+}
+
+/// IDSTS - the IDMAC's own status word, which reaches the controller's interrupt output through
+/// IDINTEN and *not* through INTMASK.
+///
+/// The second independent reason the line can be asserted, and therefore the first thing to read
+/// when a handler entry cannot be explained by RINTSTS. `muteInterrupts` leaves IDINTEN at zero so
+/// this cannot raise the line in this driver; a foreign handler entry with bits set here means
+/// something put IDINTEN back.
+pub fn dmaStatusRaw() u32 {
+ return idsts.raw();
+}
+
+/// Clear every latched event *except* this slot's SDIO card interrupt.
+///
+/// For an interrupt handler that has to lower the controller's output line without racing the
+/// card: the card interrupt is the one event the handler is being woken for, and clearing it here
+/// would drop the wakeup. Everything else - a stale command-done, a latched card-detect, an error
+/// from a transfer that has already been reported - is safe to drop on the floor, and leaving any
+/// of it latched while unmasked keeps the CLIC line high.
+pub fn clearNonSlaveInterrupts() void {
+ rintsts.writeRaw(0x0003_ffff & ~(Event.io_slot0 << state.slot));
+}
+
+/// Unmask this slot's SDIO card interrupt, i.e. let it - and after `muteInterrupts`, *only* it -
+/// reach the CLIC. RINTSTS records the event either way, so polling works without this.
+///
+/// This is the whole of the masking half of arming a waiter: `muteInterrupts` has already left
+/// every other bit of INTMASK and all of IDINTEN at zero, so `true` here makes this slot's card
+/// interrupt the single reason the controller's output can assert - which is what a
+/// level-triggered CLIC line with a one-cause handler requires.
+///
+/// The *order* around it is the part that is easy to get wrong, and it belongs to whoever owns the
+/// scheduler rather than here. `sd_host_slot_sdmmc_io_int_wait` (`sd_host_sdmmc.c:404-426`) is the
+/// reference, and it is four steps:
+///
+/// 1. `setSlaveInterruptEnabled(false)` - mask, so nothing arrives while the state is in flux.
+/// 2. `clearSlaveInterrupt()` - drop the latched edge, so a stale one is not delivered as news.
+/// 3. `slaveInterruptAsserted()` - **if true, act now and do not sleep.** The capture is a
+/// negedge, so with D1 already low step 2 has just thrown away the only edge there will be.
+/// 4. `setSlaveInterruptEnabled(true)` - unmask, and not before. Nothing can be lost between 2
+/// and 4: D1 is a level, and unmasking a bit RINTSTS has already latched asserts the line at
+/// once.
+///
+/// A handler on that line must mask again as its **unconditional first act**, on every path
+/// including the one where the cause turns out not to be its own. The line is a level and it does
+/// not lower itself.
+pub fn setSlaveInterruptEnabled(on: bool) void {
+ // Masked, and that is not decoration. `sdioDispatch` performs *this same* read-modify-write on
+ // *this same* two-bit field, from an interrupt handler, as its unconditional first act. A task
+ // interrupted between the load and the store puts back the bit the handler had just cleared -
+ // re-arming a level-triggered line with nobody left waiting on it, which is precisely how the
+ // storm gets its second chance. Two CSR instructions, and `clkrst.Guard` composes: called from
+ // inside a handler, where MIE is already clear, it leaves MIE clear.
+ //
+ // The register writes are unchanged, so the `sdio_interrupt` differential case still compares
+ // the same resulting word against `sdmmc_ll_enable_sdio_interrupt`'s.
+ const guard = intr.mask();
+ defer guard.release();
+ const m = slotBit();
+ const cur = intmask.get(sdio_int_mask);
+ intmask.modify(.{sdio_int_mask.is(if (on) cur | m else cur & ~m)});
+}
+
+/// What the controller reported the instant after this slot's card interrupt was unmasked.
+pub const Armed = struct {
+ /// The bit the store was trying to set, i.e. `slaveInterruptMask()`.
+ want: u32,
+ /// INTMASK, read back inside the same masked region as the store.
+ intmask: u32,
+ /// MINTSTS - `RINTSTS & INTMASK`, and the only word the controller's output follows. Zero here
+ /// with `stuck()` true is the normal way to enter a sleep: the mask took and nothing is latched
+ /// yet.
+ mintsts: u32,
+ /// RINTSTS, for the case where the edge landed between the unmask and the read-back.
+ rintsts: u32,
+
+ /// Did the unmask take effect?
+ pub inline fn stuck(self: Armed) bool {
+ return self.intmask & self.want != 0;
+ }
+};
+
+/// Unmask this slot's card interrupt and read the result back, both inside one masked region.
+///
+/// This exists because the opposite conclusion was drawn from diagnostics that could not support
+/// it. Every arming window on this board printed `intmask=0x00000000` and that was read as "the
+/// unmask does not stick" - but `MARK PORT_SDIO_LAPSE` prints *after* the disarm, which had just
+/// written that zero deliberately, and `interruptDiagnostics` runs on the application task, which
+/// is never inside an arming window. Neither reading could ever have shown anything else, whatever
+/// the hardware did.
+///
+/// So the claim gets an instrument instead of an argument. Nothing runs between the store and the
+/// three loads: no task, because the runtime is cooperative, and no handler, because MIE is clear.
+/// A `stuck()` of false here is a fact about this register on this die; `stuck()` true retires the
+/// hypothesis.
+pub fn armSlaveInterrupt() Armed {
+ const guard = intr.mask();
+ defer guard.release();
+ const m = slotBit();
+ intmask.modify(.{sdio_int_mask.is(intmask.get(sdio_int_mask) | m)});
+ return .{
+ .want = slaveInterruptMask(),
+ .intmask = intmask.raw(),
+ .mintsts = mintsts.raw(),
+ .rintsts = rintsts.raw(),
+ };
+}
+
+/// This slot's bit in RINTSTS/INTMASK/MINTSTS - the only interrupt cause a waiter here understands.
+pub fn slaveInterruptMask() u32 {
+ return Event.io_slot0 << state.slot;
+}
+
+/// Is the card asserting its interrupt *right now*?
+///
+/// Read from D1's pad rather than from RINTSTS, because the two answer different questions: the
+/// register says "a negedge was latched and not yet cleared", the pad says "the card is holding the
+/// line low". Only the second is safe to test before sleeping, and it is what ESP-IDF tests -
+/// `gpio_get_level(slot_ctx->io_config.d1_io) == 0` at `sd_host_sdmmc.c:413-415`.
+///
+/// Requires D1's input buffer and matrix input to be configured, which `configurePins` does.
+pub fn slaveInterruptAsserted() bool {
+ return gpio.getLevel(state.pins.d1) == 0;
+}
+
+/// The masked status word - what the controller's interrupt output is actually looking at.
+///
+/// Non-zero here and a silent CLIC means the delivery path above the controller is broken (source
+/// routing, line enable, priority, threshold, mstatus.MIE). Zero here while `interruptStatusRaw`
+/// is non-zero means the event is latched but masked, which is the normal resting state of this
+/// driver.
+pub fn interruptStatusMasked() u32 {
+ return mintsts.raw();
+}
+
+// ------------------------------------------------------------------------------- diagnostics
+
+/// `ets_printf` from the mask ROM, the same declaration `src/net/port.zig:75` makes and for the
+/// same reason: this file's only module imports are `regs`, `mmio` and its sibling HALs, and the
+/// symbol comes from the generated linker script rather than from any of them.
+extern fn ets_printf(fmt: [*:0]const u8, ...) c_int;
+
+fn note(comptime fmt: [*:0]const u8, args: anytype) void {
+ _ = @call(.auto, ets_printf, .{fmt} ++ args);
+}
+
+inline fn yesno(b: bool) u32 {
+ return @intFromBool(b);
+}
+
+/// Print the whole card-interrupt delivery chain, in the order a signal traverses it, so that one
+/// flash says which link is broken. Reads registers only: no loop, no wait, no side effect on any
+/// of the state it reports.
+///
+/// The chain has four links and each line below covers one:
+///
+/// * `SDIO_DIAG_PAD` - the card's end. `asserted=1` means D1 is low, i.e. the C6 is requesting
+/// service at this instant. `ie=0` or `in_src` not equal to D1's pad number means the
+/// controller cannot see D1 at all, and no amount of unmasking will help.
+/// * `SDIO_DIAG_TIEOFF` - the two matrix inputs with no pin on this board. `card_int_n` must read
+/// 63 (constant one, inactive) and `card_detect_n` 62 (constant zero, card present), both with
+/// `from_matrix=1`. A `card_int_n` stuck at a constant *zero* is an interrupt that is asserted
+/// before software ever runs, so the first negedge happens before anyone is watching and no
+/// second one ever comes.
+/// * `SDIO_DIAG_CTLR` - the controller's end. `latched` is RINTSTS's bit for this slot,
+/// `unmasked` is INTMASK's, and `mintsts` is the conjunction the interrupt output follows.
+/// `idsts`/`idinten` are the other, independent reason this output can be asserted.
+///
+/// **`unmasked=0` here is the resting state and is not a finding.** This function is called
+/// from an application task; the card interrupt is unmasked only inside an arming window, on
+/// the transport's own task, and is masked again by the handler or the disarm before that task
+/// yields. So an application can never observe the mask up, whatever the hardware does, and
+/// reading a zero here as "the unmask does not stick" is what cost this path a week. The
+/// register read that can answer that question is `armSlaveInterrupt`.
+/// * `SDIO_DIAG_CLIC` - delivery. An unrouted source, a clear `enabled`, a priority at or below
+/// `thresh`, or `mie=0` each mean the line exists and cannot arrive.
+pub fn interruptDiagnostics() void {
+ const m = slaveInterruptMask();
+ const rsts = rintsts.raw();
+ const imask = intmask.raw();
+ const d1 = state.pins.d1;
+ const d1_in = gpio.matrixInSource(sig.cdata1);
+ const ci = gpio.matrixInSource(sig.card_int);
+ const cdet = gpio.matrixInSource(sig.card_detect);
+
+ note("MARK SDIO_DIAG_PAD d1=gpio%u level=%u asserted=%u ie=%u in_src=%u from_matrix=%u inv=%u expect in_src=%u\r\n", .{
+ @as(u32, d1),
+ @as(u32, gpio.getLevel(d1)),
+ yesno(slaveInterruptAsserted()),
+ yesno(gpio.isInputEnabled(d1)),
+ @as(u32, d1_in.pin),
+ yesno(d1_in.from_matrix),
+ yesno(d1_in.inverted),
+ @as(u32, d1),
+ });
+ note("MARK SDIO_DIAG_TIEOFF card_int_n=%u/%u card_detect_n=%u/%u expect 63/1 and 62/1\r\n", .{
+ @as(u32, ci.pin), yesno(ci.from_matrix),
+ @as(u32, cdet.pin), yesno(cdet.from_matrix),
+ });
+ note("MARK SDIO_DIAG_CTLR slot=%u bit=0x%05x latched=%u unmasked=%u rintsts=0x%08x intmask=0x%08x mintsts=0x%08x\r\n", .{
+ @as(u32, state.slot),
+ m,
+ yesno(rsts & m != 0),
+ yesno(imask & m != 0),
+ rsts,
+ imask,
+ interruptStatusMasked(),
+ });
+ note("MARK SDIO_DIAG_CTRL ctrl=0x%08x int_enable=%u idsts=0x%08x idinten=0x%08x status=0x%08x clkena=0x%08x\r\n", .{
+ ctrl.raw(),
+ ctrl.get(int_enable),
+ idsts.raw(),
+ idinten.raw(),
+ status.raw(),
+ clkena.raw(),
+ });
+
+ const src: u32 = @intFromEnum(interrupt_source);
+ if (intr.routedLine(interrupt_source)) |line| {
+ note("MARK SDIO_DIAG_CLIC source=%u line=%u enabled=%u pending=%u trigger=%u prio=%u thresh=%u mie=%u\r\n", .{
+ src,
+ @as(u32, line),
+ yesno(intr.isEnabled(line)),
+ yesno(intr.isPending(line)),
+ @as(u32, @intFromEnum(intr.getTrigger(line))),
+ @as(u32, intr.getPriority(line)),
+ @as(u32, intr.getThreshold()),
+ yesno(intr.globalEnabled()),
+ });
+ } else {
+ note("MARK SDIO_DIAG_CLIC source=%u UNROUTED - no CLIC line can deliver this interrupt\r\n", .{src});
+ }
+}
+
+// --------------------------------------------------------------------------------- host tests
+//
+// Everything below runs on the host under `zig build test`. It covers the two things in this file
+// that are pure functions of their arguments - the command word and the CMD52/CMD53 argument
+// layouts - plus the divider table and the chunking rule. The register sequences are not testable
+// here; that is what src/oracle/sdmmc_cases.zig is for.
+
+const testing = std.testing;
+
+test "CMD52 read is the word ESP-IDF builds" {
+ // make_hw_cmd for {opcode 52, SCF_CMD_AC | SCF_RSP_R5}: response_expect (R5 is PRESENT),
+ // check_response_crc (R5 has CRC), wait_complete (not CMD0/12/11), no data. Then
+ // sd_host_slot_start_command adds use_hold_reg, card_num and start_command.
+ const w = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_2174), w);
+}
+
+test "CMD52 on slot 0 differs from slot 1 only in card_num" {
+ const s0 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 0 });
+ const s1 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA000_2174), s0);
+ try testing.expectEqual(@as(u32, 1 << 16), s0 ^ s1);
+}
+
+test "CMD53 sets data_expected, and rw only when writing" {
+ const rd = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .read, .slot = 1 });
+ const wr = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .write, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_2375), rd);
+ try testing.expectEqual(@as(u32, 0xA001_2775), wr);
+ try testing.expectEqual(@as(u32, 1 << 10), rd ^ wr);
+}
+
+test "CMD0 sends the init sequence and expects nothing back" {
+ // The only command make_hw_cmd gives send_init and denies wait_complete.
+ const w = commandWord(.{ .index = 0, .send_init = true, .wait_prvdata = false, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_8000), w);
+ try testing.expectEqual(@as(u32, 0), w & (1 << 6)); // no response expected
+}
+
+test "CMD5's response CRC is not checked" {
+ // R4 is SCF_RSP_PRESENT alone (sd_protocol_types.h:141) - the OCR response carries no valid
+ // CRC7, and checking it would fail every card.
+ const w = commandWord(.{ .index = 5, .response = .short, .check_crc = false, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_2045), w);
+ try testing.expectEqual(@as(u32, 0), w & (1 << 8));
+}
+
+test "CMD3 and CMD7 do check it" {
+ try testing.expectEqual(
+ @as(u32, 0xA001_2143),
+ commandWord(.{ .index = 3, .response = .short, .check_crc = true, .slot = 1 }),
+ );
+ try testing.expectEqual(
+ @as(u32, 0xA001_2147),
+ commandWord(.{ .index = 7, .response = .short, .check_crc = true, .slot = 1 }),
+ );
+}
+
+test "the clock update command sends nothing to the card" {
+ const w = commandWord(.{ .index = 0, .update_clock = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA021_2000), w);
+ try testing.expectEqual(@as(u32, 1 << 21), w & (1 << 21));
+ try testing.expectEqual(@as(u32, 0), w & (1 << 6));
+}
+
+test "a long response sets response_length as well as response_expect" {
+ const w = commandWord(.{ .index = 2, .response = .long, .check_crc = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 1 << 7), w & (1 << 7));
+ try testing.expectEqual(@as(u32, 1 << 6), w & (1 << 6));
+}
+
+test "every command word starts the command and uses the hold register" {
+ for ([_]Command{
+ .{ .index = 52, .response = .short, .check_crc = true },
+ .{ .index = 53, .response = .short, .check_crc = true, .data = .read },
+ .{ .index = 0, .send_init = true, .wait_prvdata = false },
+ }) |c| {
+ const w = commandWord(c);
+ try testing.expect(w & (1 << 31) != 0);
+ try testing.expect(w & (1 << 29) != 0);
+ }
+}
+
+test "CMD52 argument layout" {
+ // Read CCCR 0x00 on function 0: everything zero.
+ try testing.expectEqual(@as(u32, 0), cmd52Arg(false, 0, 0x00, false, 0));
+ // Write 0x02 to CCCR 0x02 (I/O enable) on function 0.
+ try testing.expectEqual(@as(u32, 0x8000_0402), cmd52Arg(true, 0, 0x02, false, 0x02));
+ // Function 1, address 0x1F800, data 0xAB, with the read-after-write flag.
+ const a = cmd52Arg(true, 1, 0x1F800, true, 0xAB);
+ try testing.expectEqual(@as(u32, 1), a >> 31);
+ try testing.expectEqual(@as(u32, 1), (a >> 28) & 0x7);
+ try testing.expectEqual(@as(u32, 1), (a >> 27) & 1);
+ try testing.expectEqual(@as(u32, 0x1F800), (a >> 9) & 0x1FFFF);
+ try testing.expectEqual(@as(u32, 0xAB), a & 0xFF);
+}
+
+test "CMD53 argument layout, both modes" {
+ // Block mode, function 1, address 0, incrementing, one block.
+ const blk = cmd53Arg(false, 1, 0, true, true, 1);
+ try testing.expectEqual(@as(u32, 0x1C00_0001), blk);
+ // Byte mode, function 1, fixed address, 12 bytes - the ESP-Hosted length read.
+ const byt = cmd53Arg(false, 1, 0x058, false, false, 12);
+ try testing.expectEqual(@as(u32, 0x1000_B00C), byt);
+ // Writing sets bit 31 and nothing else.
+ try testing.expectEqual(
+ @as(u32, 1) << 31,
+ cmd53Arg(true, 1, 0x058, false, false, 12) ^ byt,
+ );
+}
+
+test "byte mode encodes 512 as a count of zero" {
+ // SDIO simplified spec 5.3.1, as applied at sdmmc_io.c:351-355. The chunker never produces
+ // this case - a 512-byte request is a whole block and goes block mode - so the encoder is
+ // checked directly. ESP-Hosted can still reach it: a 512-byte read at a fixed address.
+ try testing.expectEqual(@as(u32, 0), cmd53Arg(false, 1, 0, false, true, 0) & 0x1ff);
+ const c = nextChunk(512);
+ try testing.expect(c.block_mode);
+ try testing.expectEqual(@as(u32, 512), c.len);
+ try testing.expectEqual(@as(u9, 1), c.count);
+}
+
+test "chunking splits on block boundaries and clamps to the bounce buffer" {
+ // Whole blocks, within the buffer: one block-mode command.
+ try testing.expectEqual(@as(u32, 1024), nextChunk(1024).len);
+ try testing.expect(nextChunk(1024).block_mode);
+ // More blocks than fit: clamped to bounce_len, still block mode, still whole blocks.
+ const big = nextChunk(8192);
+ try testing.expect(big.block_mode);
+ try testing.expectEqual(bounce_len, big.len);
+ try testing.expectEqual(@as(u9, bounce_len / 512), big.count);
+ // Not a block multiple: byte mode, count in bytes.
+ const odd = nextChunk(12);
+ try testing.expect(!odd.block_mode);
+ try testing.expectEqual(@as(u32, 12), odd.len);
+ try testing.expectEqual(@as(u9, 12), odd.count);
+ // Longer than one byte-mode command can carry: clamped to 512.
+ const long = nextChunk(1000);
+ try testing.expect(!long.block_mode);
+ try testing.expectEqual(@as(u32, 512), long.len);
+}
+
+test "the controller's 4-byte rule turns a 6-byte transfer into 4 then 2" {
+ // sd_trans_sdmmc.c:526-532 rejects a length that is >= 4 and not a multiple of 4 outright.
+ const first = nextChunk(6);
+ try testing.expect(!first.block_mode);
+ try testing.expectEqual(@as(u32, 4), first.len);
+ const second = nextChunk(6 - first.len);
+ try testing.expectEqual(@as(u32, 2), second.len);
+ // Under four bytes the whole thing goes in one command; that is the case the rule exempts.
+ try testing.expectEqual(@as(u32, 3), nextChunk(3).len);
+ try testing.expectEqual(@as(u32, 1), nextChunk(1).len);
+ // And every chunk a loop produces is either aligned or a final short tail.
+ var remaining: u32 = 1023;
+ var commands: u32 = 0;
+ while (remaining > 0) {
+ const c = nextChunk(remaining);
+ try testing.expect(c.len > 0);
+ try testing.expect(c.len < 4 or c.len % 4 == 0);
+ remaining -= c.len;
+ commands += 1;
+ try testing.expect(commands < 8); // 512 + 508 + 3, not an unbounded walk
+ }
+}
+
+test "divider table reproduces ESP-IDF's three named frequencies" {
+ try testing.expectEqual(Dividers{ .host = 10, .card = 20 }, dividersFor(400));
+ try testing.expectEqual(Dividers{ .host = 8, .card = 0 }, dividersFor(20_000));
+ try testing.expectEqual(Dividers{ .host = 4, .card = 0 }, dividersFor(40_000));
+}
+
+test "divider table lands on or below the requested frequency" {
+ for ([_]u32{ 400, 1_000, 5_000, 10_000, 20_000, 25_000, 40_000 }) |khz| {
+ const d = dividersFor(khz);
+ const div: u64 = @as(u64, d.host) * (if (d.card == 0) @as(u64, 1) else @as(u64, d.card) * 2);
+ const actual_khz = 160_000 / div;
+ try testing.expect(actual_khz <= khz);
+ }
+}
+
+test "the DMA region is one aligned block of exactly the documented size" {
+ try testing.expectEqual(@as(usize, 16), @sizeOf(Descriptor));
+ try testing.expectEqual(@as(usize, 64 + bounce_len), @sizeOf(DmaRegion));
+ try testing.expectEqual(@as(usize, 0), @offsetOf(DmaRegion, "desc"));
+ try testing.expectEqual(@as(usize, 64), @offsetOf(DmaRegion, "buf"));
+ // Both halves of what the one-shot cache maintenance call needs: a base on a cache line and a
+ // length that is a whole number of them. The alignment is on the variable, not on the type -
+ // `@alignOf(DmaRegion)` is 4 - so it has to be checked on the object.
+ try testing.expectEqual(@as(usize, 0), @intFromPtr(&dma) % cache_line);
+ try testing.expectEqual(@as(usize, 0), @sizeOf(DmaRegion) % cache_line);
+}
+
+test "the non-cacheable alias is a fixed offset and nothing more" {
+ var cell: u32 = 0;
+ const a = cachedAddr(&cell);
+ try testing.expectEqual(a +% @as(usize, 0x4000_0000), uncachedAddr(&cell));
+ // And it is the offset ESP-IDF uses, not one this file invented.
+ try testing.expectEqual(@as(u32, 0x4000_0000), non_cacheable_offset);
+}
+
+test "descriptor flags are the bits sdmmc_struct.h names" {
+ try testing.expectEqual(@as(u32, 1 << 2), Descriptor.last_descriptor);
+ try testing.expectEqual(@as(u32, 1 << 3), Descriptor.first_descriptor);
+ try testing.expectEqual(@as(u32, 1 << 4), Descriptor.second_address_chained);
+ try testing.expectEqual(@as(u32, 1 << 31), Descriptor.owned_by_idmac);
+ // The word a single-descriptor transfer writes.
+ const flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor |
+ Descriptor.last_descriptor | Descriptor.second_address_chained;
+ try testing.expectEqual(@as(u32, 0x8000_001C), flags);
+}
+
+test "the default interrupt mask is ESP-IDF's SDMMC_LL_EVENT_DEFAULT" {
+ // sdmmc_ll.h:64-69, expanded: CD|RESP_ERR|CMD_DONE|DATA_OVER|RCRC|DCRC|RTO|DTO|HTO|HLE|SBE|EBE
+ try testing.expectEqual(@as(u32, 0xB7CF), Event.default);
+ // and it deliberately excludes the two per-FIFO-word requests and both SDIO card interrupts.
+ try testing.expectEqual(@as(u32, 0), Event.default & (Event.txdr | Event.rxdr));
+ try testing.expectEqual(@as(u32, 0), Event.default & (Event.io_slot0 | Event.io_slot1));
+}
+
+test "the armed mask drops card detect, and nothing else" {
+ // The bit that produced an unstoppable CLIC line 21: cd latches during pin setup, nothing in
+ // the command path clears it, and while it is unmasked the controller's output never
+ // deasserts. `configureInterrupts` writes `armed`, not `default`.
+ try testing.expectEqual(@as(u32, 0xB7CE), Event.armed);
+ try testing.expectEqual(@as(u32, 0), Event.armed & Event.cd);
+ try testing.expectEqual(Event.cd, Event.default ^ Event.armed);
+ // Every event a transfer actually waits on survives the change.
+ for ([_]u32{ Event.cmd_done, Event.dto, Event.re, Event.rcrc, Event.dcrc, Event.rto, Event.drto, Event.hto, Event.hle, Event.sbe, Event.ebe }) |e| {
+ try testing.expect(Event.armed & e != 0);
+ }
+}
diff --git a/src/hal/systimer.zig b/src/hal/systimer.zig
new file mode 100644
index 0000000..dc316bf
--- /dev/null
+++ b/src/hal/systimer.zig
@@ -0,0 +1,151 @@
+//! SYSTIMER: two 52-bit counters on a fixed clock, plus three comparators each.
+//!
+//! This is the most useful peripheral on the chip for bring-up work and the cheapest to trust. Its
+//! source is fixed - XTAL at 40 MHz, divided to 16 MHz (`clk_tree_defs.h:196-198`) - so unlike the
+//! CPU cycle counter its rate does not move when the clock tree is reconfigured, and unlike the
+//! timer groups it needs no divider arithmetic and no pads.
+//!
+//! Reading it is a **sequence**, not a load, and that is the interesting part:
+//!
+//! write UNIT0_UPDATE = 1 -> ask the peripheral to latch its counter
+//! poll UNIT0_VALUE_VALID -> wait for the latch
+//! read VALUE_HI, then VALUE_LO -> read the latched pair
+//!
+//! Skip the handshake and you read a value that is being incremented underneath you: the low word
+//! can wrap between the two loads, so `hi` belongs to one instant and `lo` to the next, and the
+//! result jumps backwards by 2^32 ticks about once every 268 seconds at 16 MHz. A register
+//! snapshot taken after either version looks identical - which is exactly why the differential
+//! harness records the *write trace* as well as the final state.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const clkrst = @import("clkrst.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+/// Ticks per second. XTAL/2.5 = 16 MHz, fixed: `SYSTIMER_CLK_SRC_XTAL` with the divider ESP-IDF
+/// programs in `systimer_hal_init`. Not derived from the CPU clock, which on this board is whatever
+/// the bootloader left (measured ~90 MHz, not the 360 the part is rated for).
+pub const hz: u32 = 16_000_000;
+
+const conf = Reg.at(regs.SYSTIMER_CONF_REG);
+const clk_en = Field.of(regs.SYSTIMER_CLK_EN_S, regs.SYSTIMER_CLK_EN_V);
+
+/// The two counter units. `unit_op` holds the update/valid handshake bits, `value_hi`/`value_lo` the
+/// latched result. Strides are derived from consecutive macros, not assumed.
+const unit_op = mmio.RegArray(regs.SYSTIMER_UNIT0_OP_REG, regs.SYSTIMER_UNIT1_OP_REG, 2);
+const unit_value_hi = mmio.RegArray(regs.SYSTIMER_UNIT0_VALUE_HI_REG, regs.SYSTIMER_UNIT1_VALUE_HI_REG, 2);
+const unit_value_lo = mmio.RegArray(regs.SYSTIMER_UNIT0_VALUE_LO_REG, regs.SYSTIMER_UNIT1_VALUE_LO_REG, 2);
+
+// The per-unit fields split into two groups, and the split is not obvious from the names.
+//
+// `update` and `valid` live in a *per-unit* register (UNIT0_OP_REG, UNIT1_OP_REG) and therefore sit
+// at the same bit in each - asserted below, so indexing the register is enough.
+//
+// `work_en` is different: both units' enables live in the *shared* SYSTIMER_CONF_REG, at bits 30 and
+// 29 respectively. A first draft of this file used unit 0's field for both, which would have enabled
+// the wrong counter and left the requested one dead; the comptime assert caught it before it ever
+// reached the chip. Hence a per-unit lookup rather than one constant.
+const update = Field.of(regs.SYSTIMER_TIMER_UNIT0_UPDATE_S, regs.SYSTIMER_TIMER_UNIT0_UPDATE_V);
+const valid = Field.of(regs.SYSTIMER_TIMER_UNIT0_VALUE_VALID_S, regs.SYSTIMER_TIMER_UNIT0_VALUE_VALID_V);
+const value_hi = Field.of(regs.SYSTIMER_TIMER_UNIT0_VALUE_HI_S, regs.SYSTIMER_TIMER_UNIT0_VALUE_HI_V);
+
+comptime {
+ const update1 = Field.of(regs.SYSTIMER_TIMER_UNIT1_UPDATE_S, regs.SYSTIMER_TIMER_UNIT1_UPDATE_V);
+ const valid1 = Field.of(regs.SYSTIMER_TIMER_UNIT1_VALUE_VALID_S, regs.SYSTIMER_TIMER_UNIT1_VALUE_VALID_V);
+ if (update1.shift != update.shift or valid1.shift != valid.shift)
+ @compileError("the systimer units' OP registers disagree on bit positions; index per unit");
+ // The other half of the same story: these two MUST differ, because they share a register.
+ if (workEn(.unit0).shift == workEn(.unit1).shift)
+ @compileError("both work_en fields claim the same bit of SYSTIMER_CONF; one macro is wrong");
+}
+
+inline fn workEn(comptime unit: Unit) Field {
+ return switch (unit) {
+ .unit0 => Field.of(regs.SYSTIMER_TIMER_UNIT0_WORK_EN_S, regs.SYSTIMER_TIMER_UNIT0_WORK_EN_V),
+ .unit1 => Field.of(regs.SYSTIMER_TIMER_UNIT1_WORK_EN_S, regs.SYSTIMER_TIMER_UNIT1_WORK_EN_V),
+ };
+}
+
+pub const Unit = enum(u1) { unit0 = 0, unit1 = 1 };
+
+/// The counter's own clock gate, inside the peripheral and separate from the bus clock gate in
+/// HP_SYS_CLKRST.
+pub fn setEnabled(on: bool) void {
+ conf.modify(.{clk_en.is(@intFromBool(on))});
+}
+
+pub fn setUnitEnabled(comptime unit: Unit, on: bool) void {
+ conf.modify(.{workEn(unit).is(@intFromBool(on))});
+}
+
+/// Bring the peripheral up: bus clock and reset through CLKRST, then its internal gate and unit.
+///
+/// Deliberately does *not* reprogram the clock source or divider. The bootloader has already set
+/// those, ESP-IDF's own `systimer_hal_init` would set them the same way, and re-running that on a
+/// live counter makes the timebase jump - which would corrupt any measurement taken across the call.
+pub fn init() void {
+ clkrst.setClockEnabled(.systimer, true);
+ setEnabled(true);
+ setUnitEnabled(.unit0, true);
+}
+
+/// The 52-bit counter, latched through the update/valid handshake.
+///
+/// Returns null if the peripheral does not acknowledge within `spins` reads, rather than spinning
+/// forever: a systimer whose clock is gated off never sets `valid`, and hanging in a HAL call with
+/// no output is the worst possible way to report that.
+pub fn read(unit: Unit) ?u64 {
+ const i: u32 = @intFromEnum(unit);
+ const op = unit_op.at(i);
+
+ // Ask for a snapshot, and clear the previous handshake in the same store.
+ //
+ // This has to be a read-modify-write, and a whole-word `write` is a bug. UPDATE is bit 30 and
+ // `WT`, so writing it as a single store looks right - but VALUE_VALID is bit 29 of the same word
+ // and is `R/SS/WTC`, write-1-to-clear (systimer_reg.h). A whole-word store writes 0 there, which
+ // is the no-op for a W1C bit, so the valid flag from the *previous* snapshot is never cleared:
+ // after one successful read it stays set forever, the poll below exits immediately on a stale
+ // flag, and the HI/LO pair that follows can straddle two different snapshots - precisely the
+ // tearing this handshake exists to prevent.
+ //
+ // ESP-IDF gets this right by accident of its idiom: `systimer_ll_counter_snapshot` assigns a
+ // bitfield of a `volatile` union, which compiles to a 32-bit read-modify-write that writes bit
+ // 29 back as 1 whenever it read 1, clearing it and re-arming in one store. This does the same
+ // thing deliberately.
+ //
+ // The register differential cannot see this: once any snapshot has completed, UNIT0_OP reads
+ // 0x2000_0000 under either version.
+ op.writeRaw(op.raw() | update.mask());
+
+ var spins: u32 = 0;
+ while (op.get(valid) == 0) {
+ spins += 1;
+ if (spins > 10_000) return null;
+ }
+
+ // Order matters less than the latch does - both words are frozen now - but read high first to
+ // match ESP-IDF's LL, so the write/read trace lines up under differential test.
+ const hi: u64 = unit_value_hi.at(i).get(value_hi);
+ const lo: u64 = unit_value_lo.at(i).raw();
+ return (hi << 32) | lo;
+}
+
+/// Microseconds since the counter started, from the 16 MHz tick.
+pub fn micros(unit: Unit) ?u64 {
+ const ticks = read(unit) orelse return null;
+ return ticks / (hz / 1_000_000);
+}
+
+/// Busy-wait. Uses the counter rather than the CPU cycle count, so the delay is right regardless of
+/// what the CPU clock happens to be.
+pub fn delayMicros(us: u32) void {
+ const start = read(.unit0) orelse return;
+ const target = start + @as(u64, us) * (hz / 1_000_000);
+ while (true) {
+ const now = read(.unit0) orelse return;
+ if (now >= target) return;
+ }
+}
diff --git a/src/hal/timg.zig b/src/hal/timg.zig
new file mode 100644
index 0000000..f669387
--- /dev/null
+++ b/src/hal/timg.zig
@@ -0,0 +1,513 @@
+//! The timer groups: TIMG0 and TIMG1, each two general-purpose 54-bit timers plus one MWDT.
+//!
+//! Three unrelated functions share one register block (timg_ll.h:7 says so in as many words):
+//! the general-purpose timers, the main watchdog, and RTC clock calibration. Only the first two are
+//! here; calibration belongs to the clock tree, and ETM and interrupts are deliberately absent.
+//!
+//! Four things about this block cost real care, all of them taken from ESP-IDF's LL rather than
+//! guessed at:
+//!
+//! **Reading the counter is a sequence, not a load** (timer_ll.h:248-269). The counter lives in a
+//! different clock domain from the register file, so its value only appears in `TxLO`/`TxHI` after a
+//! software capture:
+//!
+//! write TIMG_TxUPDATE = 1 -> ask for a capture
+//! poll until TIMG_Tx_UPDATE == 0 -> the hardware clears it when the pair is latched
+//! read TxHI, then TxLO -> 22 bits + 32 bits = the 54-bit count
+//!
+//! Note the polarity: unlike SYSTIMER, which sets a separate `VALUE_VALID` bit, this peripheral
+//! *clears the request bit* to acknowledge. Waiting for it to become 1 hangs forever; not waiting at
+//! all returns whatever the last capture left, which for a never-captured timer is 0 and therefore
+//! looks like a stopped timer rather than like a bug.
+//!
+//! **The watchdog registers are write-protected, and the key is the reset value** (mwdt_ll.h:231-244
+//! and timer_group_reg.h, TIMG_WDT_WKEY: "If the register contains a different value than its reset
+//! value, write protection is enabled", default 1356348065 = 0x50D83AA1). So "unlock" means writing
+//! the key back, and "lock" means writing anything else - IDF writes 0. A watchdog register write
+//! made while locked is silently dropped, which is the failure mode this file's API shape exists to
+//! prevent: every MWDT operation is a method on the `Watchdog` handle returned by `unlock`, and
+//! there is no way to reach one without holding it:
+//!
+//! const wdt = timg.unlock(.timg1);
+//! defer wdt.release();
+//! wdt.setStage(.stage0, 2_000_000, .reset_system);
+//!
+//! **Watchdog configuration is committed asynchronously.** Every write to WDTCONFIG0-5 has to be
+//! followed by `WDT_CONF_UPDATE_EN` (mwdt_ll.h:122, and again after every other config write), which
+//! is a write-to-trigger bit. The exception is `WDT_EN` itself: `mwdt_ll_enable`/`_disable`
+//! (mwdt_ll.h:61-77) do *not* pulse it, so neither does `setEnabled` - matching IDF exactly matters
+//! more here than consistency, because the differential harness compares the resulting word.
+//!
+//! **Do not resurrect a watchdog you are not feeding.** TIMG0 hosts MWDT0, which this image's
+//! bootloader has already disabled, and `TIMG_WDT_FLASHBOOT_MOD_EN` defaults to 1 and runs the
+//! watchdog *independently of* `WDT_EN` (mwdt_ll.h:186-188). Resetting a timer group therefore
+//! re-arms flash-boot protection and reboots the board a moment later with nothing on the console to
+//! explain it; `clkrst.resetPeripheral` clears the bit as part of the reset for exactly this reason
+//! (its `clears_flashboot` flag), which is why nothing in this file pulses a reset bit itself.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const clkrst = @import("clkrst.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+/// TIMG_LL_INST_NUM (timg_ll.h:20).
+pub const group_count = 2;
+/// TIMG_LL_GPTIMERS_PER_INST (timg_ll.h:23). Two per group on the P4, unlike the C-series parts.
+pub const timers_per_group = 2;
+/// TIMER_LL_COUNTER_BIT_WIDTH (timer_ll.h:25). 32 bits in `TxLO` plus 22 in `TxHI`.
+pub const counter_bits = 54;
+
+pub const Group = enum(u1) { timg0 = 0, timg1 = 1 };
+pub const Timer = enum(u1) { t0 = 0, t1 = 1 };
+
+// -------------------------------------------------------------------------------- addressing
+//
+// The macros are indexed two different ways at once and neither is derivable from the other:
+// `TIMG_T0CONFIG_REG(i)` takes the *group*, while the *timer* is baked into the macro name
+// (`T0CONFIG` vs `T1CONFIG`). Rather than duplicate every accessor per timer, the timer index is
+// turned into a stride - but a stride assumed is a stride that eventually writes into the next
+// register, so both strides are checked at comptime against the macros for the other instance.
+
+const group_stride = mmio.addr(regs.TIMG_T0CONFIG_REG(1)) - mmio.addr(regs.TIMG_T0CONFIG_REG(0));
+const timer_stride = mmio.addr(regs.TIMG_T1CONFIG_REG(0)) - mmio.addr(regs.TIMG_T0CONFIG_REG(0));
+
+// Absolute addresses of group 0 / timer 0's registers. Every other (group, timer) is these plus a
+// multiple of the two strides.
+const a_config = mmio.addr(regs.TIMG_T0CONFIG_REG(0));
+const a_lo = mmio.addr(regs.TIMG_T0LO_REG(0));
+const a_hi = mmio.addr(regs.TIMG_T0HI_REG(0));
+const a_update = mmio.addr(regs.TIMG_T0UPDATE_REG(0));
+const a_alarm_lo = mmio.addr(regs.TIMG_T0ALARMLO_REG(0));
+const a_alarm_hi = mmio.addr(regs.TIMG_T0ALARMHI_REG(0));
+const a_load_lo = mmio.addr(regs.TIMG_T0LOADLO_REG(0));
+const a_load_hi = mmio.addr(regs.TIMG_T0LOADHI_REG(0));
+const a_load = mmio.addr(regs.TIMG_T0LOAD_REG(0));
+
+comptime {
+ // The timer sub-block is contiguous and uniform - assert it, per register, rather than trust
+ // that 0x24 happens to be right for all nine.
+ const pairs = .{
+ .{ a_config, mmio.addr(regs.TIMG_T1CONFIG_REG(0)) },
+ .{ a_lo, mmio.addr(regs.TIMG_T1LO_REG(0)) },
+ .{ a_hi, mmio.addr(regs.TIMG_T1HI_REG(0)) },
+ .{ a_update, mmio.addr(regs.TIMG_T1UPDATE_REG(0)) },
+ .{ a_alarm_lo, mmio.addr(regs.TIMG_T1ALARMLO_REG(0)) },
+ .{ a_alarm_hi, mmio.addr(regs.TIMG_T1ALARMHI_REG(0)) },
+ .{ a_load_lo, mmio.addr(regs.TIMG_T1LOADLO_REG(0)) },
+ .{ a_load_hi, mmio.addr(regs.TIMG_T1LOADHI_REG(0)) },
+ .{ a_load, mmio.addr(regs.TIMG_T1LOAD_REG(0)) },
+ };
+ for (pairs) |p| {
+ if (p[1] - p[0] != timer_stride) @compileError(
+ "the two timers' registers are not a uniform stride apart; index them per timer",
+ );
+ }
+ // And the group stride is the same for a register other than CONFIG.
+ if (mmio.addr(regs.TIMG_T0LO_REG(1)) - a_lo != group_stride)
+ @compileError("the two timer groups are not a uniform stride apart");
+
+ // T0's and T1's *fields* sit at the same bit positions in their respective registers, which is
+ // what makes one set of Field constants enough. If a future register set moves one of them,
+ // this stops the build instead of writing the divider into the alarm enable.
+ const t1_divider = Field.of(regs.TIMG_T1_DIVIDER_S, regs.TIMG_T1_DIVIDER_V);
+ const t1_en = Field.of(regs.TIMG_T1_EN_S, regs.TIMG_T1_EN_V);
+ const t1_update = Field.of(regs.TIMG_T1_UPDATE_S, regs.TIMG_T1_UPDATE_V);
+ const t1_hi = Field.of(regs.TIMG_T1_HI_S, regs.TIMG_T1_HI_V);
+ if (t1_divider.shift != divider.shift or t1_divider.width != divider.width or
+ t1_en.shift != counter_en.shift or t1_update.shift != update.shift or
+ t1_hi.width != count_hi.width)
+ @compileError("timer 0 and timer 1 disagree on field positions; look up fields per timer");
+}
+
+inline fn tReg(comptime a0: u32, g: Group, t: Timer) Reg {
+ return Reg.atAddress(a0 +
+ group_stride * @as(u32, @intFromEnum(g)) +
+ timer_stride * @as(u32, @intFromEnum(t)));
+}
+
+/// `a0` is not comptime: the stage-timeout registers are picked by a runtime `Stage`
+/// (`stageHoldAddr`), and every other caller passes a constant that folds anyway.
+inline fn gReg(a0: u32, g: Group) Reg {
+ return Reg.atAddress(a0 + group_stride * @as(u32, @intFromEnum(g)));
+}
+
+// TxCONFIG fields. `divcnt_rst` is write-to-trigger; the rest are plain R/W.
+const alarm_en = Field.of(regs.TIMG_T0_ALARM_EN_S, regs.TIMG_T0_ALARM_EN_V);
+const divcnt_rst = Field.of(regs.TIMG_T0_DIVCNT_RST_S, regs.TIMG_T0_DIVCNT_RST_V);
+const divider = Field.of(regs.TIMG_T0_DIVIDER_S, regs.TIMG_T0_DIVIDER_V);
+const autoreload = Field.of(regs.TIMG_T0_AUTORELOAD_S, regs.TIMG_T0_AUTORELOAD_V);
+const increase = Field.of(regs.TIMG_T0_INCREASE_S, regs.TIMG_T0_INCREASE_V);
+const counter_en = Field.of(regs.TIMG_T0_EN_S, regs.TIMG_T0_EN_V);
+const update = Field.of(regs.TIMG_T0_UPDATE_S, regs.TIMG_T0_UPDATE_V);
+const count_hi = Field.of(regs.TIMG_T0_HI_S, regs.TIMG_T0_HI_V);
+const alarm_value_hi = Field.of(regs.TIMG_T0_ALARM_HI_S, regs.TIMG_T0_ALARM_HI_V);
+const load_value_hi = Field.of(regs.TIMG_T0_LOAD_HI_S, regs.TIMG_T0_LOAD_HI_V);
+
+// ------------------------------------------------------------------------------ timer clocks
+//
+// The timers' function clock is selected and gated in HP_SYS_CLKRST, not in the timer group: group 0
+// in PERI_CLK_CTRL20 and group 1 in PERI_CLK_CTRL21 (timer_ll.h:117-129, :146-160). Two shared
+// registers, so both operations take the interrupt guard - the same read-modify-write hazard
+// `clkrst` exists for.
+
+const peri_clk_ctrl20 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL20_REG);
+const peri_clk_ctrl21 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL21_REG);
+
+/// The three function clocks a GP timer can run from, with the encodings from
+/// `timer_ll_set_clock_source` (timer_ll.h:100-116). The numbering is not the enum order anyone
+/// would pick: XTAL is 0, RC_FAST is 1, PLL_F80M is 2.
+pub const ClockSource = enum(u2) {
+ xtal = 0,
+ rc_fast = 1,
+ pll_f80m = 2,
+};
+
+/// Where the group/timer's source-select and gate fields live. Both are in one word per group, and
+/// the bit positions differ per timer, so this is a genuine per-instance lookup rather than a stride.
+const TimerClock = struct {
+ reg: Reg,
+ src_sel: Field,
+ clk_en: Field,
+};
+
+inline fn timerClock(comptime g: Group, comptime t: Timer) TimerClock {
+ return switch (g) {
+ .timg0 => switch (t) {
+ .t0 => .{
+ .reg = peri_clk_ctrl20,
+ .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_SRC_SEL_V),
+ .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_CLK_EN_V),
+ },
+ .t1 => .{
+ .reg = peri_clk_ctrl20,
+ .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_SRC_SEL_V),
+ .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_CLK_EN_V),
+ },
+ },
+ .timg1 => switch (t) {
+ .t0 => .{
+ .reg = peri_clk_ctrl21,
+ .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_SRC_SEL_V),
+ .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_CLK_EN_V),
+ },
+ .t1 => .{
+ .reg = peri_clk_ctrl21,
+ .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_SRC_SEL_V),
+ .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_CLK_EN_V),
+ },
+ },
+ };
+}
+
+/// Select a timer's function clock. Comptime instance because the field pairing really does differ
+/// per (group, timer) - four different bit positions in two registers.
+pub fn setClockSource(comptime g: Group, comptime t: Timer, src: ClockSource) void {
+ const c = comptime timerClock(g, t);
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ c.reg.modify(.{c.src_sel.is(@intFromEnum(src))});
+}
+
+/// The timer's function-clock gate, distinct from the group's bus clock in `clkrst`. Defaults to 1
+/// at power-on (hp_sys_clkrst_reg.h: REG_TIMERGRP0_T0_CLK_EN default 1).
+pub fn setClockEnabled(comptime g: Group, comptime t: Timer, on: bool) void {
+ const c = comptime timerClock(g, t);
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ c.reg.modify(.{c.clk_en.is(@intFromBool(on))});
+}
+
+// ------------------------------------------------------------------------ general purpose timer
+
+pub const Direction = enum { up, down };
+
+/// Prescaler on the function clock. 2 is the smallest the hardware accepts and 65536 the largest,
+/// encoded as 0 (timer_ll.h:191-199). The divider counter is reset in a second store afterwards,
+/// exactly as IDF does it: without that the new divider only takes effect after the old one's
+/// current period ends, so the first tick after a change is the wrong length.
+pub fn setDivider(g: Group, t: Timer, div: u32) void {
+ std.debug.assert(div >= 2 and div <= 65536);
+ const cfg = tReg(a_config, g, t);
+ cfg.modify(.{divider.is(if (div >= 65536) 0 else div)});
+ cfg.modify(.{divcnt_rst.is(1)});
+}
+
+pub fn setDirection(g: Group, t: Timer, dir: Direction) void {
+ tReg(a_config, g, t).modify(.{increase.is(@intFromBool(dir == .up))});
+}
+
+/// Reload the counter from `TxLOADLO`/`TxLOADHI` automatically on every alarm.
+pub fn setAutoReload(g: Group, t: Timer, on: bool) void {
+ tReg(a_config, g, t).modify(.{autoreload.is(@intFromBool(on))});
+}
+
+pub fn setCounterEnabled(g: Group, t: Timer, on: bool) void {
+ tReg(a_config, g, t).modify(.{counter_en.is(@intFromBool(on))});
+}
+
+pub fn setAlarmEnabled(g: Group, t: Timer, on: bool) void {
+ tReg(a_config, g, t).modify(.{alarm_en.is(@intFromBool(on))});
+}
+
+/// The 54-bit alarm value. Low word first would be equally correct - the comparator only sees the
+/// pair - but IDF writes high then low (timer_ll.h:279-283) and matching its order keeps the write
+/// trace comparable.
+pub fn setAlarmValue(g: Group, t: Timer, value: u64) void {
+ tReg(a_alarm_hi, g, t).modify(.{alarm_value_hi.is(@truncate(value >> 32))});
+ tReg(a_alarm_lo, g, t).writeRaw(@truncate(value));
+}
+
+/// The value a reload puts into the counter, whether triggered by `load` or by an auto-reload.
+pub fn setLoadValue(g: Group, t: Timer, value: u64) void {
+ tReg(a_load_hi, g, t).modify(.{load_value_hi.is(@truncate(value >> 32))});
+ tReg(a_load_lo, g, t).writeRaw(@truncate(value));
+}
+
+pub fn getLoadValue(g: Group, t: Timer) u64 {
+ const hi: u64 = tReg(a_load_hi, g, t).get(load_value_hi);
+ return (hi << 32) | tReg(a_load_lo, g, t).raw();
+}
+
+/// Copy the load value into the counter now. `TIMG_TxLOAD_REG` is a whole-word write-to-trigger
+/// register: the value written is irrelevant, so this is a bare store rather than a field write.
+pub fn load(g: Group, t: Timer) void {
+ tReg(a_load, g, t).writeRaw(1);
+}
+
+/// The counter, through the capture handshake described at the top of this file.
+///
+/// Returns null rather than spinning forever if the peripheral never acknowledges: with the group's
+/// bus clock gated off, or the timer's function clock gated off, `UPDATE` never clears, and hanging
+/// inside a HAL call with no output is the worst possible way to report that. The bound is the same
+/// 10,000 reads `systimer.read` uses.
+pub fn read(g: Group, t: Timer) ?u64 {
+ const upd = tReg(a_update, g, t);
+
+ // Ask for a capture. IDF assigns to the struct bitfield, which is a read-modify-write of a word
+ // whose only other bits are reserved, so `modify` is both the honest operation and the one that
+ // produces the same store.
+ upd.modify(.{update.is(1)});
+
+ var spins: u32 = 0;
+ while (upd.get(update) != 0) {
+ spins += 1;
+ if (spins > 10_000) return null;
+ }
+
+ const hi: u64 = tReg(a_hi, g, t).get(count_hi);
+ return (hi << 32) | tReg(a_lo, g, t).raw();
+}
+
+// ----------------------------------------------------------------------------------- watchdog
+
+const a_wdtconfig0 = mmio.addr(regs.TIMG_WDTCONFIG0_REG(0));
+const a_wdtconfig1 = mmio.addr(regs.TIMG_WDTCONFIG1_REG(0));
+const a_wdtconfig2 = mmio.addr(regs.TIMG_WDTCONFIG2_REG(0));
+const a_wdtconfig3 = mmio.addr(regs.TIMG_WDTCONFIG3_REG(0));
+const a_wdtconfig4 = mmio.addr(regs.TIMG_WDTCONFIG4_REG(0));
+const a_wdtconfig5 = mmio.addr(regs.TIMG_WDTCONFIG5_REG(0));
+const a_wdtfeed = mmio.addr(regs.TIMG_WDTFEED_REG(0));
+const a_wdtwprotect = mmio.addr(regs.TIMG_WDTWPROTECT_REG(0));
+
+const wdt_en = Field.of(regs.TIMG_WDT_EN_S, regs.TIMG_WDT_EN_V);
+const wdt_conf_update_en = Field.of(regs.TIMG_WDT_CONF_UPDATE_EN_S, regs.TIMG_WDT_CONF_UPDATE_EN_V);
+const wdt_flashboot_mod_en = Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V);
+const wdt_cpu_reset_length = Field.of(regs.TIMG_WDT_CPU_RESET_LENGTH_S, regs.TIMG_WDT_CPU_RESET_LENGTH_V);
+const wdt_sys_reset_length = Field.of(regs.TIMG_WDT_SYS_RESET_LENGTH_S, regs.TIMG_WDT_SYS_RESET_LENGTH_V);
+const wdt_clk_prescale = Field.of(regs.TIMG_WDT_CLK_PRESCALE_S, regs.TIMG_WDT_CLK_PRESCALE_V);
+const wdt_divcnt_rst = Field.of(regs.TIMG_WDT_DIVCNT_RST_S, regs.TIMG_WDT_DIVCNT_RST_V);
+
+/// The write-protect key, and also `TIMG_WDT_WKEY`'s reset value: protection is on whenever the
+/// register holds anything *else* (timer_group_reg.h, TIMG_WDT_WKEY, default 1356348065). IDF's
+/// `mwdt_ll_write_protect_disable` writes this exact constant (mwdt_ll.h:243).
+pub const wkey: u32 = 0x50D8_3AA1;
+
+/// What IDF writes to re-enable protection (mwdt_ll.h:233). Any non-key value would do; using the
+/// same one keeps the register comparable against IDF's.
+const wkey_locked: u32 = 0;
+
+// The headers carry no reset-value macro to check `wkey` against - `TIMG_WDT_WKEY_V` is the field
+// mask, 0xffffffff - so the constant is copied from the two places that state it: the register
+// description's "default: 1356348065" and mwdt_ll.h:243's 0x50D83AA1. The unit test at the end of
+// this file pins those two against each other, which is the only check available without a chip.
+
+/// MWDT stages, each with its own timeout and its own action. Stage 0 fires first; a stage that is
+/// not fed escalates to the next.
+pub const Stage = enum(u2) { stage0 = 0, stage1 = 1, stage2 = 2, stage3 = 3 };
+
+/// What a stage does when it expires (mwdt_ll.h:23-26).
+pub const Action = enum(u2) {
+ off = 0,
+ interrupt = 1,
+ reset_cpu = 2,
+ reset_system = 3,
+};
+
+/// Length of the reset pulse a `reset_cpu`/`reset_system` stage asserts (mwdt_ll.h:28-35).
+pub const ResetLength = enum(u3) {
+ ns_100 = 0,
+ ns_200 = 1,
+ ns_300 = 2,
+ ns_400 = 3,
+ ns_500 = 4,
+ ns_800 = 5,
+ us_1_6 = 6,
+ us_3_2 = 7,
+};
+
+/// A group's MWDT with write protection lifted, and the only way to reach an MWDT operation:
+///
+/// const wdt = timg.unlock(.timg1);
+/// defer wdt.release();
+/// wdt.setStage(.stage0, ticks, .reset_system);
+///
+/// The handle exists because a watchdog register write made while protection is on is silently
+/// dropped - no fault, no status bit, just a watchdog that keeps its old timeout - and that is not a
+/// mistake worth making twice.
+pub const Watchdog = struct {
+ group: Group,
+
+ /// Re-enable write protection. Not idempotent-with-`unlock` in the composable sense that
+ /// `clkrst.Guard` is: the hardware has one key register and no nesting count, so an inner
+ /// `release` really does lock an outer caller out. There is nothing in this HAL that nests.
+ pub inline fn release(self: Watchdog) void {
+ gReg(a_wdtwprotect, self.group).writeRaw(wkey_locked);
+ }
+
+ /// WDTCONFIG0-5 are shadowed; the hardware only takes them at a `CONF_UPDATE_EN` pulse
+ /// (mwdt_ll.h:121-122). Write-to-trigger, so this is a single deliberate store.
+ inline fn commit(self: Watchdog) void {
+ gReg(a_wdtconfig0, self.group).modify(.{wdt_conf_update_en.is(1)});
+ }
+
+ inline fn config0(self: Watchdog) Reg {
+ return gReg(a_wdtconfig0, self.group);
+ }
+
+ /// The stage's action bits and its timeout live in different registers - the action in
+ /// WDTCONFIG0, the timeout in WDTCONFIG2+stage - which is why this takes both at once
+ /// (mwdt_ll.h:98-123). `timeout` is in MWDT clock cycles, i.e. after the prescaler.
+ pub fn setStage(self: Watchdog, stage: Stage, timeout: u32, action: Action) void {
+ self.config0().modify(.{stageAction(stage).is(@intFromEnum(action))});
+ gReg(stageHoldAddr(stage), self.group).writeRaw(timeout);
+ self.commit();
+ }
+
+ /// Turn one stage off without disturbing its timeout (mwdt_ll.h:131-152).
+ pub fn disableStage(self: Watchdog, stage: Stage) void {
+ self.config0().modify(.{stageAction(stage).is(@intFromEnum(Action.off))});
+ self.commit();
+ }
+
+ pub fn getStageTimeout(self: Watchdog, stage: Stage) u32 {
+ return gReg(stageHoldAddr(stage), self.group).raw();
+ }
+
+ /// Prescaler from the MWDT's source clock (XTAL on this chip - mwdt_ll.h:273-283 asserts it and
+ /// selects nothing). 1 to 65535; IDF's default is 20000, which gives 500 ticks/us
+ /// (mwdt_ll.h:20).
+ pub fn setPrescaler(self: Watchdog, prescaler: u32) void {
+ std.debug.assert(prescaler >= 1 and prescaler <= 0xffff);
+ gReg(a_wdtconfig1, self.group).modify(.{wdt_clk_prescale.is(prescaler)});
+ self.commit();
+ }
+
+ pub fn setCpuResetLength(self: Watchdog, len: ResetLength) void {
+ self.config0().modify(.{wdt_cpu_reset_length.is(@intFromEnum(len))});
+ self.commit();
+ }
+
+ pub fn setSysResetLength(self: Watchdog, len: ResetLength) void {
+ self.config0().modify(.{wdt_sys_reset_length.is(@intFromEnum(len))});
+ self.commit();
+ }
+
+ /// Flash-boot protection: a second, independent way for this watchdog to run. It ignores
+ /// `WDT_EN` entirely (mwdt_ll.h:186-188), it defaults to 1, and a group reset re-arms it - so
+ /// clearing it is part of every sane bring-up, and `clkrst.resetPeripheral` does it.
+ pub fn setFlashbootEnabled(self: Watchdog, on: bool) void {
+ self.config0().modify(.{wdt_flashboot_mod_en.is(@intFromBool(on))});
+ self.commit();
+ }
+
+ /// Start or stop the watchdog. No `CONF_UPDATE_EN` pulse: `mwdt_ll_enable` and `_disable`
+ /// (mwdt_ll.h:61-77) do not, so neither does this. Disabling does *not* stop flash-boot mode.
+ pub fn setEnabled(self: Watchdog, on: bool) void {
+ self.config0().modify(.{wdt_en.is(@intFromBool(on))});
+ }
+
+ pub fn isEnabled(self: Watchdog) bool {
+ return self.config0().get(wdt_en) == 1;
+ }
+
+ /// Reset the count and the stage. `TIMG_WDTFEED_REG` is a whole-word write-to-trigger register,
+ /// so the value is irrelevant (mwdt_ll.h:219-222).
+ pub fn feed(self: Watchdog) void {
+ gReg(a_wdtfeed, self.group).writeRaw(1);
+ }
+
+ /// Reset the watchdog's clock divider counter. Write-to-trigger, in WDTCONFIG1 alongside the
+ /// prescaler.
+ pub fn resetDividerCount(self: Watchdog) void {
+ gReg(a_wdtconfig1, self.group).modify(.{wdt_divcnt_rst.is(1)});
+ self.commit();
+ }
+};
+
+/// Lift write protection and hand back the only handle that can touch the MWDT.
+pub fn unlock(g: Group) Watchdog {
+ gReg(a_wdtwprotect, g).writeRaw(wkey);
+ return .{ .group = g };
+}
+
+/// Feed a watchdog, protection dance included. The one MWDT operation that is worth a shortcut,
+/// because it is the one called from a loop.
+pub fn feed(g: Group) void {
+ const wdt = unlock(g);
+ defer wdt.release();
+ wdt.feed();
+}
+
+/// True if write protection is currently on, i.e. the key register holds something other than the
+/// key. Reads the register, so it reports the hardware rather than what this module last wrote.
+pub fn isWriteProtected(g: Group) bool {
+ return gReg(a_wdtwprotect, g).raw() != wkey;
+}
+
+inline fn stageAction(stage: Stage) Field {
+ // Stage 0 is at the *top* of the word (bits 30:29) and stage 3 at 24:23, i.e. the stages run
+ // downwards through the register. The four are a uniform 2 bits apart, but in the reverse of
+ // the obvious direction, so they are looked up rather than computed.
+ return switch (stage) {
+ .stage0 => Field.of(regs.TIMG_WDT_STG0_S, regs.TIMG_WDT_STG0_V),
+ .stage1 => Field.of(regs.TIMG_WDT_STG1_S, regs.TIMG_WDT_STG1_V),
+ .stage2 => Field.of(regs.TIMG_WDT_STG2_S, regs.TIMG_WDT_STG2_V),
+ .stage3 => Field.of(regs.TIMG_WDT_STG3_S, regs.TIMG_WDT_STG3_V),
+ };
+}
+
+inline fn stageHoldAddr(stage: Stage) u32 {
+ // WDTCONFIG2 holds stage 0's timeout and WDTCONFIG5 stage 3's; the mapping is off by two and
+ // there is no macro that says so, so it comes from mwdt_ll.h:100-116.
+ return switch (stage) {
+ .stage0 => a_wdtconfig2,
+ .stage1 => a_wdtconfig3,
+ .stage2 => a_wdtconfig4,
+ .stage3 => a_wdtconfig5,
+ };
+}
+
+test "the two strides are the documented ones" {
+ // 0x1000 between groups (timer_group_reg.h:14, REG_TIMG_BASE) and 0x24 between the two timers
+ // of a group. Both are asserted against the macros at comptime above; this pins the numbers so
+ // a header change shows up as a failing test with a value in it, not only as a compile error.
+ try std.testing.expectEqual(@as(u32, 0x1000), group_stride);
+ try std.testing.expectEqual(@as(u32, 0x24), timer_stride);
+}
+
+test "the write-protect key is the register's reset value" {
+ try std.testing.expectEqual(@as(u32, 1_356_348_065), wkey);
+}
diff --git a/src/hal/uart.zig b/src/hal/uart.zig
new file mode 100644
index 0000000..7c891a2
--- /dev/null
+++ b/src/hal/uart.zig
@@ -0,0 +1,622 @@
+//! The HP UART controllers: UART0-4.
+//!
+//! Out of scope on purpose: UHCI/DMA, RS485, IrDA, hardware and software flow control, the wakeup
+//! machinery, and LP_UART (which is a different block behind a different clock tree, not an
+//! instance of this one).
+//!
+//! Three things about this peripheral cost real debugging time, and all three are structural rather
+//! than incidental:
+//!
+//! **Half the configuration registers are shadowed.** The registers whose macro name ends `_SYNC` -
+//! UART_CLKDIV_SYNC, UART_CONF0_SYNC, and a dozen more - are not the live configuration. A write
+//! lands in a shadow that the core clock domain ignores until UART_REG_UPDATE is set, at which
+//! point the hardware copies the shadow across and clears the bit itself. Reads come back from the
+//! shadow, so a read-modify-write composes correctly and a read-back proves nothing about what the
+//! transmitter is currently using. Every mutator here therefore ends in `update()`, which is
+//! exactly what ESP-IDF does: `uart_ll_update` (uart_ll.h:85-89) sets the bit and spins on it, and
+//! every `*_sync` writer in that file calls it (set_stop_bits at uart_ll.h:793, set_parity at 828,
+//! set_data_bit_num at 1023, set_loop_back at 1439, the FIFO resets at 735 and 750). Omitting it
+//! does not fail loudly: the register reads back as asked and the wire keeps the old setting.
+//!
+//! **Reading offset 0x000 pops the RX FIFO.** `UART_FIFO_REG`'s only field is annotated `RO` in
+//! uart_reg.h:18 and that annotation is wrong in the way that matters - the read is the pop. A
+//! generic "snapshot the block" loop therefore eats received bytes, which is why the differential
+//! harness carries a per-peripheral deny-list of offsets. Writes to the same address push a byte,
+//! and must be full 32-bit stores: a byte store on this bus is a read-modify-write, so it would pop
+//! a byte in order to push one (uart_ll.h:716-724 says so and is the reason `pushByte` uses
+//! `writeRaw`).
+//!
+//! **UART0 is the console.** Resetting it clears UART_CLKDIV, the console turns to garbage
+//! mid-sentence and the board dies on a watchdog reset with nothing readable to explain it. That was
+//! measured on this board. Nothing here resets UART0 implicitly, `reset()` refuses instance 0, and
+//! the differential suite uses UART1.
+//!
+//! The clock path is two dividers in series and they live in different blocks: HP_SYS_CLKRST holds
+//! the integer pre-divider (`REG_UARTn_SCLK_DIV_NUM`) and the source select, the UART itself holds
+//! the 12.4 fixed-point divider. `setBaudrate` drives both, because neither alone spans the range.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const gpio = @import("gpio.zig");
+const clkrst = @import("clkrst.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+/// UART0-4. LP_UART (ESP-IDF's port 5) is a separate peripheral and not modelled here.
+pub const count = 5;
+
+/// SOC_UART_FIFO_LEN, soc_caps.h:655. Both directions; the TX count register reports how many bytes
+/// are queued, so free space is this minus that.
+pub const fifo_len = 128;
+
+// ------------------------------------------------------------------------------ register blocks
+// One 0x1000-byte block per instance (soc.h:20, `REG_UART_BASE(i) = DR_REG_UART_BASE + i*0x1000`).
+// Every register is reached through a RegArray so the stride is checked against the headers rather
+// than assumed, and a wrong instance index is a bounds assert rather than a write into UART2.
+
+fn regArray(comptime offset: u32) type {
+ return mmio.RegArray(
+ regs.DR_REG_UART0_BASE + offset,
+ regs.DR_REG_UART0_BASE + 0x1000 + offset,
+ count,
+ );
+}
+
+const fifo = regArray(0x00);
+const clkdiv_sync = regArray(0x14);
+const status = regArray(0x1c);
+const conf0_sync = regArray(0x20);
+const clk_conf = regArray(0x88);
+const reg_update = regArray(0x98);
+
+// CLKDIV_SYNC: a 12.4 fixed-point divider, with the fraction not adjacent to the integer part.
+const clkdiv = Field.of(regs.UART_CLKDIV_S, regs.UART_CLKDIV_V);
+const clkdiv_frag = Field.of(regs.UART_CLKDIV_FRAG_S, regs.UART_CLKDIV_FRAG_V);
+
+// CONF0_SYNC: the data format, the FIFO resets and the loopback switch all share this word, which is
+// why every one of them is a read-modify-write and not a `write`.
+const parity = Field.of(regs.UART_PARITY_S, regs.UART_PARITY_V);
+const parity_en = Field.of(regs.UART_PARITY_EN_S, regs.UART_PARITY_EN_V);
+const bit_num = Field.of(regs.UART_BIT_NUM_S, regs.UART_BIT_NUM_V);
+const stop_bit_num = Field.of(regs.UART_STOP_BIT_NUM_S, regs.UART_STOP_BIT_NUM_V);
+const loopback = Field.of(regs.UART_LOOPBACK_S, regs.UART_LOOPBACK_V);
+const rxfifo_rst = Field.of(regs.UART_RXFIFO_RST_S, regs.UART_RXFIFO_RST_V);
+const txfifo_rst = Field.of(regs.UART_TXFIFO_RST_S, regs.UART_TXFIFO_RST_V);
+
+// STATUS: live counters, so read-only and never worth comparing between two runs.
+const rxfifo_cnt = Field.of(regs.UART_RXFIFO_CNT_S, regs.UART_RXFIFO_CNT_V);
+const txfifo_cnt = Field.of(regs.UART_TXFIFO_CNT_S, regs.UART_TXFIFO_CNT_V);
+
+const tx_sclk_en = Field.of(regs.UART_TX_SCLK_EN_S, regs.UART_TX_SCLK_EN_V);
+const rx_sclk_en = Field.of(regs.UART_RX_SCLK_EN_S, regs.UART_RX_SCLK_EN_V);
+
+/// UART_REG_UPDATE, the commit bit for the whole `_SYNC` family. `R/W/SC`: the hardware clears it
+/// when the copy is done.
+const reg_update_bit = Field.of(regs.UART_REG_UPDATE_S, regs.UART_REG_UPDATE_V);
+
+// -------------------------------------------------------------------------------- clock control
+// The source select and the integer pre-divider are one register apart, and not in the register the
+// names suggest: for UARTn the select is in PERI_CLK_CTRL(110+n) and the pre-divider is in
+// PERI_CLK_CTRL(111+n). That is not a typo in this file - uart_ll.h:463-475 writes
+// `peri_clk_ctrl110.reg_uart0_clk_src_sel` while uart_ll.h:558-568 writes
+// `peri_clk_ctrl111.reg_uart0_sclk_div_num`, so ctrl111 holds UART0's divider *and* UART1's select.
+
+const peri_clk_ctrl = mmio.RegArray(
+ regs.HP_SYS_CLKRST_PERI_CLK_CTRL110_REG,
+ regs.HP_SYS_CLKRST_PERI_CLK_CTRL111_REG,
+ 6, // ctrl110..ctrl115: five selects and five dividers, overlapping by one
+);
+
+// All five instances place these fields at the same shifts in their respective registers
+// (hp_sys_clkrst_reg.h: every REG_UARTn_CLK_SRC_SEL_S is 24, every REG_UARTn_SCLK_DIV_NUM_S is 0,
+// every REG_UARTn_CLK_EN_S is 26), so one macro triple each describes all of them.
+const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_V);
+const sclk_div_num = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_V);
+const sclk_en = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_V);
+
+/// The three clock sources an HP UART can take, with the encoding from uart_ll.h:447-461.
+pub const ClockSource = enum(u2) {
+ /// The 40 MHz crystal. The only source whose frequency is exact, which is why it is the default
+ /// for anything that has to interoperate.
+ xtal = 0,
+ /// RC_FAST, the always-on oscillator. Nominally 20 MHz and uncalibrated - it varies with
+ /// temperature and part, so a baud rate derived from `nominalHz` here is approximate.
+ rtc = 1,
+ /// A fixed 80 MHz tap off the system PLL. Needed for the high rates: the 12-bit integer divider
+ /// runs out below about 5 kBd from XTAL.
+ pll_f80m = 2,
+
+ /// The nominal frequency to hand `setBaudrate`. Nominal is exact for `xtal` and `pll_f80m` and a
+ /// datasheet typical for `rtc`; the real clock tree can be reconfigured, so a caller that has
+ /// changed it must pass its own number instead.
+ pub fn nominalHz(self: ClockSource) u32 {
+ return switch (self) {
+ .xtal => 40_000_000,
+ .rtc => 20_000_000,
+ .pll_f80m => 80_000_000,
+ };
+ }
+};
+
+pub const WordLength = enum(u2) {
+ // uart_types.h:57-60. The encoding is (bits - 5), which is why it starts at zero.
+ bits5 = 0,
+ bits6 = 1,
+ bits7 = 2,
+ bits8 = 3,
+};
+
+pub const StopBits = enum(u2) {
+ // uart_types.h:68-70. There is no encoding for zero stop bits, so the enum starts at 1 and 0 is
+ // reserved by the hardware.
+ one = 1,
+ one_and_half = 2,
+ two = 3,
+};
+
+pub const Parity = enum(u2) {
+ // uart_types.h:78-80: bit 1 is "parity enabled", bit 0 is odd/even. `disable` is 0, so the
+ // odd/even bit is not part of it - see `setParity` for why that matters.
+ disable = 0,
+ even = 2,
+ odd = 3,
+};
+
+/// One UART instance. A value type holding nothing but the index, so it costs nothing at runtime and
+/// every register access folds to a constant address when the index is known.
+pub const Uart = struct {
+ num: u8,
+
+ pub fn init(num: u8) Uart {
+ std.debug.assert(num < count);
+ return .{ .num = num };
+ }
+
+ // ----------------------------------------------------------------------------- the commit bit
+
+ /// Copy the `_SYNC` shadow registers into the core clock domain and wait for the hardware to
+ /// acknowledge by clearing the bit (uart_ll.h:85-89).
+ ///
+ /// Bounded, where ESP-IDF's `while (hw->reg_update.reg_update);` is not: a UART whose core clock
+ /// is gated off never clears the bit, and on a board with no debugger an infinite spin is
+ /// indistinguishable from a crash. 4096 spins is several thousand times the observed cost of a
+ /// commit, which takes a handful of core-clock cycles. Returns false rather than panicking so a
+ /// caller can report the peripheral instead of losing the console.
+ pub fn update(self: Uart) bool {
+ const r = reg_update.at(self.num);
+ r.modify(.{reg_update_bit.is(1)});
+ return r.waitFor(reg_update_bit, 0, 4096);
+ }
+
+ // ---------------------------------------------------------------------------- clocks and reset
+
+ /// Reset the block. Refuses UART0.
+ ///
+ /// UART0 carries this board's console. A reset clears UART_CLKDIV to its power-on 694, the
+ /// console's output becomes garbage part-way through whatever it was printing, and the board
+ /// takes a watchdog reset a moment later - measured, not theorised. There is no "and then put
+ /// the divider back" version of this that is safe, because the damage is done between the two
+ /// stores.
+ pub fn reset(self: Uart) void {
+ std.debug.assert(self.num != 0);
+ switch (self.num) {
+ 1 => clkrst.resetPeripheral(.uart1),
+ 2 => clkrst.resetPeripheral(.uart2),
+ 3 => clkrst.resetPeripheral(.uart3),
+ 4 => clkrst.resetPeripheral(.uart4),
+ else => unreachable,
+ }
+ }
+
+ /// The core (baud-generating) clock, as distinct from the APB bus clock that
+ /// `clkrst.setClockEnabled` handles. Both are needed: the bus clock makes the registers
+ /// answer, this one makes the shift registers move - and `update()` is one of the things that
+ /// stops working without it.
+ ///
+ /// Two gates in two blocks, per uart_ll.h:379-397: HP_SYS_CLKRST's per-instance `CLK_EN`, which
+ /// sits in the *select* register PERI_CLK_CTRL(110+n) and not the divider one next to it, and
+ /// the UART's own TX and RX enables in UART_CLK_CONF. Interrupts are masked over the first
+ /// because PERI_CLK_CTRL is shared with unrelated peripherals.
+ pub fn setCoreClockEnabled(self: Uart, on: bool) void {
+ const v: u32 = @intFromBool(on);
+ {
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ self.selectReg().modify(.{sclk_en.is(v)});
+ }
+ clk_conf.at(self.num).modify(.{ tx_sclk_en.is(v), rx_sclk_en.is(v) });
+ }
+
+ /// Select the clock the baud generator divides down. Read-modify-write of a register shared with
+ /// other peripherals, so interrupts are masked (uart_ll.h:477-481 makes the equivalent point by
+ /// refusing to compile outside `PERIPH_RCC_ATOMIC`).
+ pub fn setClockSource(self: Uart, src: ClockSource) void {
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ self.selectReg().modify(.{clk_src_sel.is(@intFromEnum(src))});
+ }
+
+ pub fn clockSource(self: Uart) ClockSource {
+ // Encoding 3 is not defined; IDF's getter (uart_ll.h:509-524) maps `default` to RTC, so
+ // reporting the same thing keeps a round-trip through both implementations consistent.
+ return switch (self.selectReg().get(clk_src_sel)) {
+ 0 => .xtal,
+ 2 => .pll_f80m,
+ else => .rtc,
+ };
+ }
+
+ /// PERI_CLK_CTRL(110+n): where this instance's source select and core clock gate live.
+ inline fn selectReg(self: Uart) Reg {
+ return peri_clk_ctrl.at(self.num);
+ }
+
+ /// PERI_CLK_CTRL(111+n): where this instance's integer pre-divider lives. One register above
+ /// the select, which is the trap this pair of accessors exists to contain.
+ inline fn dividerReg(self: Uart) Reg {
+ return peri_clk_ctrl.at(self.num + 1);
+ }
+
+ // ------------------------------------------------------------------------------------- baud
+
+ /// The two dividers a baud rate decomposes into, computed exactly as
+ /// `_uart_ll_set_baudrate` (uart_ll.h:532-588) does.
+ pub const Divider = struct {
+ /// HP_SYS_CLKRST's integer pre-divider, 1-256. Stored as `sclk - 1` in an 8-bit field.
+ sclk: u32,
+ /// The UART's own divider, integer part, 12 bits.
+ int: u32,
+ /// The UART's own divider, sixteenths.
+ frag: u32,
+ };
+
+ /// Decompose a baud rate, or fail if the hardware cannot express it.
+ ///
+ /// The arithmetic, line by line against uart_ll.h:
+ ///
+ /// 541 max_div = UART_CLKDIV_V = 0xfff - the UART divider's integer part is 12 bits
+ /// 542 sclk = ceil(sclk_freq / (max_div * baud)) the smallest pre-divide that brings
+ /// the remaining ratio inside 12 bits
+ /// 545 reject sclk == 0 or sclk > 256 256 = SCLK_DIV_NUM_V + 1
+ /// 549 clk_div = (sclk_freq << 4) / (baud * sclk) the ratio in sixteenths
+ /// 551 int = clk_div >> 4
+ /// 552 frag = clk_div & 0xf
+ /// 555+ the field written is sclk - 1
+ ///
+ /// The `<< 4` is IDF's fixed-point scale, not a fudge: CLKDIV_FRAG is a count of sixteenths of a
+ /// source-clock period added to every bit time, so `clk_div` is the exact ratio rounded down to
+ /// 1/16 of a tick. Two deliberate departures from the C, neither of which changes a result:
+ ///
+ /// * The `ceil` denominator is 64-bit here as it is there (uart_ll.h:542 casts `max_div` to
+ /// `uint64_t`), and `sclk_freq << 4` is *also* computed in 64 bits. In C that shift is
+ /// `uint32_t` and overflows above 268.4 MHz; no P4 UART source is anywhere near that (the
+ /// fastest is PLL_F80M at 80 MHz), so the two agree on every reachable input while this one
+ /// has no undefined case.
+ /// * `baud == 0` returns null rather than false-with-registers-untouched; same outcome, but the
+ /// caller cannot ignore it by accident.
+ pub fn divider(baud: u32, sclk_freq: u32) ?Divider {
+ if (baud == 0) return null;
+ const max_div: u64 = clkdiv.max(); // UART_CLKDIV_V
+ const denom = max_div * baud;
+ const sclk: u64 = (@as(u64, sclk_freq) + denom - 1) / denom;
+ if (sclk == 0 or sclk > @as(u64, sclk_div_num.max()) + 1) return null;
+ const clk_div: u64 = (@as(u64, sclk_freq) << 4) / (@as(u64, baud) * sclk);
+ return .{
+ .sclk = @intCast(sclk),
+ .int = @intCast(clk_div >> 4),
+ .frag = @intCast(clk_div & 0xf),
+ };
+ }
+
+ /// Program a baud rate. Returns false, having touched nothing, if it is unreachable from this
+ /// source frequency.
+ ///
+ /// Store order follows uart_ll.h:550-576 exactly - integer part, fraction, pre-divider, commit -
+ /// because the intermediate states are visible to the transmitter of a UART that is already
+ /// running, and because a write-trace comparison against IDF would otherwise differ on ordering
+ /// while agreeing on the final registers. The two CLKDIV_SYNC stores are separate for the same
+ /// reason: IDF's two bitfield assignments are two read-modify-writes of that word.
+ pub fn setBaudrate(self: Uart, baud: u32, sclk_freq: u32) bool {
+ const d = divider(baud, sclk_freq) orelse return false;
+ const div = clkdiv_sync.at(self.num);
+ div.modify(.{clkdiv.is(d.int)});
+ div.modify(.{clkdiv_frag.is(d.frag)});
+ {
+ const guard = clkrst.maskInterrupts();
+ defer guard.release();
+ self.dividerReg().modify(.{sclk_div_num.is(d.sclk - 1)});
+ }
+ _ = self.update();
+ return true;
+ }
+
+ /// The baud rate the registers currently describe, by inverting the above
+ /// (uart_ll.h:590-615). Integer division both ways, so this is not exactly the value passed to
+ /// `setBaudrate` - it is what the hardware will actually produce, which is the more useful
+ /// number.
+ pub fn baudrate(self: Uart, sclk_freq: u32) u32 {
+ const div = clkdiv_sync.at(self.num).raw();
+ const int = (div >> clkdiv.shift) & clkdiv.unshiftedMask();
+ const frag = (div >> clkdiv_frag.shift) & clkdiv_frag.unshiftedMask();
+ const sclk = self.dividerReg().get(sclk_div_num) + 1;
+ const ticks = ((@as(u64, int) << 4) | frag) * sclk;
+ if (ticks == 0) return 0;
+ return @intCast((@as(u64, sclk_freq) << 4) / ticks);
+ }
+
+ // ------------------------------------------------------------------------------ data format
+
+ /// uart_ll.h:1020-1024.
+ pub fn setWordLength(self: Uart, w: WordLength) void {
+ conf0_sync.at(self.num).modify(.{bit_num.is(@intFromEnum(w))});
+ _ = self.update();
+ }
+
+ /// uart_ll.h:790-794.
+ pub fn setStopBits(self: Uart, s: StopBits) void {
+ conf0_sync.at(self.num).modify(.{stop_bit_num.is(@intFromEnum(s))});
+ _ = self.update();
+ }
+
+ /// uart_ll.h:817-832.
+ ///
+ /// Note what IDF does *not* do: disabling parity leaves UART_PARITY - the odd/even select bit -
+ /// at whatever it was, because the value 0 for "disabled" carries no odd/even information and
+ /// writing bit 0 of it would be writing a zero the caller never asked for. So `.disable` clears
+ /// `parity_en` only. Reproduced here because otherwise a differential run diverges by one bit
+ /// after any sequence that sets odd parity and then disables it.
+ pub fn setParity(self: Uart, p: Parity) void {
+ const c = conf0_sync.at(self.num);
+ const v = @intFromEnum(p);
+ if (p != .disable) c.modify(.{parity.is(v & 1)});
+ c.modify(.{parity_en.is((v >> 1) & 1)});
+ _ = self.update();
+ }
+
+ /// All three format fields, in IDF's order. Three commits rather than one, matching what
+ /// calling IDF's three setters does: the format of a UART mid-transmission is not atomic on
+ /// this hardware either way, and diverging here would be a difference with no benefit.
+ pub fn setFormat(self: Uart, w: WordLength, p: Parity, s: StopBits) void {
+ self.setWordLength(w);
+ self.setParity(p);
+ self.setStopBits(s);
+ }
+
+ pub fn wordLength(self: Uart) WordLength {
+ return @enumFromInt(conf0_sync.at(self.num).get(bit_num));
+ }
+
+ pub fn stopBits(self: Uart) StopBits {
+ // Encoding 0 is not a legal stop-bit count. The hardware's reset value is 1, and nothing
+ // here can write 0, so an out-of-range read means the block is unclocked or was reset
+ // under us - reported as `one` rather than an illegal enum value, which would be UB.
+ return switch (conf0_sync.at(self.num).get(stop_bit_num)) {
+ 2 => .one_and_half,
+ 3 => .two,
+ else => .one,
+ };
+ }
+
+ /// uart_ll.h:834-841: parity is only meaningful when enabled, so the odd/even bit is not
+ /// reported unless it is.
+ pub fn parityMode(self: Uart) Parity {
+ const c = conf0_sync.at(self.num).raw();
+ if ((c >> parity_en.shift) & 1 == 0) return .disable;
+ return if ((c >> parity.shift) & 1 == 1) .odd else .even;
+ }
+
+ // ------------------------------------------------------------------------------------- FIFO
+
+ /// Bytes waiting in the RX FIFO (uart_ll.h:763-766).
+ pub fn rxCount(self: Uart) u32 {
+ return status.at(self.num).get(rxfifo_cnt);
+ }
+
+ /// Bytes queued in the TX FIFO.
+ pub fn txCount(self: Uart) u32 {
+ return status.at(self.num).get(txfifo_cnt);
+ }
+
+ /// Free space in the TX FIFO (uart_ll.h:775-780: the total, minus what is queued).
+ pub fn txFree(self: Uart) u32 {
+ return fifo_len - self.txCount();
+ }
+
+ /// Push one byte. A full 32-bit store, because a narrower one becomes a read-modify-write on
+ /// this bus and the read would pop a received byte (uart_ll.h:716-724).
+ pub inline fn pushByte(self: Uart, byte: u8) void {
+ fifo.at(self.num).writeRaw(byte);
+ }
+
+ /// Pop one byte. The read *is* the pop - see this file's header on why offset 0x000 is on the
+ /// differential harness's no-read list.
+ pub inline fn popByte(self: Uart) u8 {
+ return @truncate(fifo.at(self.num).raw());
+ }
+
+ /// Discard everything received. Assert, commit, deassert, commit: `rxfifo_rst` lives in a
+ /// shadow register, so without the commits the hardware never sees either edge
+ /// (uart_ll.h:733-739).
+ pub fn resetRxFifo(self: Uart) void {
+ const c = conf0_sync.at(self.num);
+ c.modify(.{rxfifo_rst.is(1)});
+ _ = self.update();
+ c.modify(.{rxfifo_rst.is(0)});
+ _ = self.update();
+ }
+
+ /// uart_ll.h:748-754. Same shape, and the same reason for it.
+ pub fn resetTxFifo(self: Uart) void {
+ const c = conf0_sync.at(self.num);
+ c.modify(.{txfifo_rst.is(1)});
+ _ = self.update();
+ c.modify(.{txfifo_rst.is(0)});
+ _ = self.update();
+ }
+
+ // --------------------------------------------------------------------------------- loopback
+
+ /// Tie TX back to RX inside the block (uart_ll.h:1437-1441). The pads are not involved, which
+ /// makes it the only way to exercise a UART end to end with nothing wired to the board - it is
+ /// how the FIFO and format paths can be tested at all here.
+ pub fn setLoopback(self: Uart, on: bool) void {
+ conf0_sync.at(self.num).modify(.{loopback.is(@intFromBool(on))});
+ _ = self.update();
+ }
+
+ pub fn loopbackEnabled(self: Uart) bool {
+ return conf0_sync.at(self.num).get(loopback) == 1;
+ }
+
+ // ------------------------------------------------------------------------------ pin routing
+
+ /// This instance's TX signal index in the GPIO matrix. The names in IDF's map are
+ /// `UARTn_TXD_PAD_OUT_IDX` (gpio_sig_map.h:28-52) and they are consecutive in steps of three,
+ /// but the step is not relied on: each is named.
+ pub fn txSignal(self: Uart) u32 {
+ return switch (self.num) {
+ 0 => regs.UART0_TXD_PAD_OUT_IDX,
+ 1 => regs.UART1_TXD_PAD_OUT_IDX,
+ 2 => regs.UART2_TXD_PAD_OUT_IDX,
+ 3 => regs.UART3_TXD_PAD_OUT_IDX,
+ 4 => regs.UART4_TXD_PAD_OUT_IDX,
+ else => unreachable,
+ };
+ }
+
+ /// This instance's RX signal index. Numerically equal to the TX one - the matrix's input and
+ /// output signal spaces are separate namespaces that happen to share indices for a duplex
+ /// peripheral - which is exactly why routing RX with `matrixOut` silently does nothing useful.
+ pub fn rxSignal(self: Uart) u32 {
+ return switch (self.num) {
+ 0 => regs.UART0_RXD_PAD_IN_IDX,
+ 1 => regs.UART1_RXD_PAD_IN_IDX,
+ 2 => regs.UART2_RXD_PAD_IN_IDX,
+ 3 => regs.UART3_RXD_PAD_IN_IDX,
+ 4 => regs.UART4_RXD_PAD_IN_IDX,
+ else => unreachable,
+ };
+ }
+
+ /// Route TX to a pad through the GPIO matrix.
+ pub fn routeTx(self: Uart, pin: u8) void {
+ gpio.matrixOut(pin, self.txSignal());
+ }
+
+ /// Route a pad to RX through the GPIO matrix, and enable that pad's input buffer - without
+ /// which the routed signal reads as a constant and the UART receives nothing, which is the
+ /// single most common way this goes wrong.
+ pub fn routeRx(self: Uart, pin: u8) void {
+ gpio.setInputEnable(pin, true);
+ gpio.matrixIn(pin, self.rxSignal());
+ }
+
+ // --------------------------------------------------------------------------------- transfers
+
+ /// Send every byte, blocking until each fits. Bounded only by the FIFO draining, which always
+ /// progresses while the core clock is on - so unlike a blocking *read* this cannot wait on an
+ /// event that may never happen.
+ pub fn write(self: Uart, bytes: []const u8) void {
+ for (bytes) |b| {
+ while (self.txFree() == 0) {}
+ self.pushByte(b);
+ }
+ }
+
+ /// Drain up to `buf.len` received bytes and report how many there were. Does not block.
+ ///
+ /// Deliberately not blocking: nothing on the other end of a UART is obliged to send, so a
+ /// blocking read is an unbounded wait, and there is no timer in this HAL's dependency set to
+ /// bound it with. A caller that wants to wait writes the loop, and owns the decision about what
+ /// to do when the bytes never come.
+ pub fn read(self: Uart, buf: []u8) usize {
+ var n: usize = 0;
+ const available = self.rxCount();
+ while (n < buf.len and n < available) : (n += 1) buf[n] = self.popByte();
+ return n;
+ }
+
+ /// Whether the transmitter has finished: nothing queued in the FIFO.
+ ///
+ /// Not the same as "the last bit is on the wire" - the shift register still holds up to one
+ /// character after the FIFO empties. UART_FSM_STATUS reports that, and this HAL does not model
+ /// it, so a caller about to cut the clock or reconfigure the format must allow for one more
+ /// character time.
+ pub fn txIdle(self: Uart) bool {
+ return self.txCount() == 0;
+ }
+};
+
+// ------------------------------------------------------------------------------------ host tests
+// The divider arithmetic is the only part of this file that can be checked without the chip, and it
+// is the part most worth checking: every value below is IDF's formula evaluated by hand, so a
+// transcription error in `divider` fails here rather than as a garbled console.
+
+test "40 MHz XTAL, 115200 Bd: one source tick, 12.4 divider does the work" {
+ // 40e6/(4095*115200) = 0.085 -> ceil = 1. clk_div = (40e6<<4)/115200 = 5555 (5555.55 floored).
+ // 5555 = 347*16 + 3.
+ const d = Uart.divider(115200, 40_000_000).?;
+ try std.testing.expectEqual(@as(u32, 1), d.sclk);
+ try std.testing.expectEqual(@as(u32, 347), d.int);
+ try std.testing.expectEqual(@as(u32, 3), d.frag);
+ // 347 + 3/16 = 347.1875 ticks per bit -> 115,213 Bd in real arithmetic, and 115,211 as the
+ // hardware's own truncating inverse reports it (see the round-trip test): 0.01% fast either way.
+}
+
+test "80 MHz PLL, 115200 Bd: the fraction differs from the XTAL case, which is the point of it" {
+ // 80e6/(4095*115200) = 0.17 -> 1. clk_div = (80e6<<4)/115200 = 11111 = 694*16 + 7.
+ const d = Uart.divider(115200, 80_000_000).?;
+ try std.testing.expectEqual(@as(u32, 1), d.sclk);
+ try std.testing.expectEqual(@as(u32, 694), d.int);
+ try std.testing.expectEqual(@as(u32, 7), d.frag);
+}
+
+test "a rate low enough to need the pre-divider" {
+ // 300 Bd from 40 MHz: 40e6/300 = 133,333 ticks per bit, far past 12 bits.
+ // ceil(40e6/(4095*300)) = ceil(32.6) = 33. clk_div = (40e6<<4)/(300*33) = 64,646 = 4040*16 + 6.
+ const d = Uart.divider(300, 40_000_000).?;
+ try std.testing.expectEqual(@as(u32, 33), d.sclk);
+ try std.testing.expectEqual(@as(u32, 4040), d.int);
+ try std.testing.expectEqual(@as(u32, 6), d.frag);
+ try std.testing.expect(d.int <= 0xfff);
+ try std.testing.expect(d.sclk <= 256);
+}
+
+test "unreachable rates are rejected rather than rounded" {
+ // Zero is IDF's explicit early return (uart_ll.h:538).
+ try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(0, 40_000_000));
+ // 10 Bd from 40 MHz needs a pre-divide of ceil(40e6/40950) = 977, past the 8-bit field's 256.
+ try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(10, 40_000_000));
+}
+
+test "the sclk == 0 rejection is unreachable except from a zero source frequency" {
+ // Worth pinning down, because the obvious reading of uart_ll.h:545 is wrong. `sclk` is a
+ // *ceiling*, so for any non-zero source frequency it is at least 1 - asking for 4 MBd from a
+ // 1 kHz clock does NOT fail here, it yields sclk = 1 and a divider of zero, and IDF programs
+ // that just as happily. The only input that trips the branch is sclk_freq == 0.
+ const absurd = Uart.divider(4_000_000, 1000).?;
+ try std.testing.expectEqual(@as(u32, 1), absurd.sclk);
+ try std.testing.expectEqual(@as(u32, 0), absurd.int);
+ try std.testing.expectEqual(@as(u32, 0), absurd.frag);
+ try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(115200, 0));
+}
+
+test "the pre-divider field stores sclk - 1, so the reachable rates stop at 256 ticks" {
+ // The boundary IDF checks at uart_ll.h:545: sclk may be 256 because the field holds sclk-1.
+ // From 40 MHz the last rate inside it is 39 Bd, at a pre-divide of 251; 38 Bd needs 257.
+ const ok = Uart.divider(39, 40_000_000).?;
+ try std.testing.expectEqual(@as(u32, 251), ok.sclk);
+ try std.testing.expectEqual(@as(u32, 4086), ok.int);
+ try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(38, 40_000_000));
+}
+
+test "the divider round-trips through the baud rate the hardware will really produce" {
+ // What `baudrate()` computes, without a chip: the inverse of the same arithmetic.
+ const d = Uart.divider(115200, 40_000_000).?;
+ const ticks = ((@as(u64, d.int) << 4) | d.frag) * d.sclk;
+ const actual: u32 = @intCast((@as(u64, 40_000_000) << 4) / ticks);
+ // 347 + 3/16 = 347.1875 ticks per bit, and 40e6*16/5555 truncates to 115,211 Bd: 0.01% fast.
+ try std.testing.expectEqual(@as(u32, 115_211), actual);
+}