diff options
| author | Gabriel Schneider <[email protected]> | 2026-08-25 12:40:53 -0300 |
|---|---|---|
| committer | Gabriel Schneider <[email protected]> | 2026-08-25 12:46:51 -0300 |
| commit | f5f8068fac59b4f16046c2022c2fc7c7e447ef4c (patch) | |
| tree | 2731a3ed4e51cae09e184e25778eded5fc37d1f5 /src/hal | |
| download | esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.tar.gz esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.zip | |
zig-p4: pure-Zig ESP32-P4 toolchain
build.zig generates the linker script and drives Zig's own LLD; tools/image.zig
turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask
ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker.
src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers;
src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip;
src/oracle/ diffs this HAL against ESP-IDF's on the die.
Diffstat (limited to 'src/hal')
| -rw-r--r-- | src/hal/clkrst.zig | 265 | ||||
| -rw-r--r-- | src/hal/gpio.zig | 471 | ||||
| -rw-r--r-- | src/hal/i2c.zig | 1091 | ||||
| -rw-r--r-- | src/hal/intr.zig | 965 | ||||
| -rw-r--r-- | src/hal/ledc.zig | 587 | ||||
| -rw-r--r-- | src/hal/rwdt.zig | 123 | ||||
| -rw-r--r-- | src/hal/sdmmc.zig | 2002 | ||||
| -rw-r--r-- | src/hal/systimer.zig | 151 | ||||
| -rw-r--r-- | src/hal/timg.zig | 513 | ||||
| -rw-r--r-- | src/hal/uart.zig | 622 |
10 files changed, 6790 insertions, 0 deletions
diff --git a/src/hal/clkrst.zig b/src/hal/clkrst.zig new file mode 100644 index 0000000..89a27ef --- /dev/null +++ b/src/hal/clkrst.zig @@ -0,0 +1,265 @@ +//! Peripheral clock gates and resets: HP_SYS_CLKRST. +//! +//! Two things about this block are counter-intuitive on the ESP32-P4, and both were found by +//! reading ESP-IDF rather than by assuming: +//! +//! **Peripheral clocks are already on.** `esp_system/port/soc/esp32p4/clk.c:200` says so in as many +//! words - "All peripheral clocks are default enabled after chip is powered on" - and the reset +//! values in `hp_sys_clkrst_reg.h` agree: REG_UART0_APB_CLK_EN, REG_TIMERGRP0_APB_CLK_EN, +//! REG_SYSTIMER_APB_CLK_EN and REG_IOMUX_APB_CLK_EN all default to 1, with their RST_EN bits at 0. +//! An image that boots from the stock second-stage bootloader never runs `esp_perip_clk_init`, so it +//! inherits those defaults. So this file is not a prerequisite for touching a peripheral; it is what +//! you need to *re*-initialise one, and to reach the few blocks that really are gated off (TWAI is +//! the notable one: REG_TWAI0_APB_CLK_EN defaults to 0). +//! +//! **The hazard is atomicity, not gating.** Every gate and reset bit for the whole chip lives in a +//! handful of shared registers, so `enable(.uart0)` is a read-modify-write of a word that also holds +//! the gate for unrelated peripherals. ESP-IDF makes unguarded calls impossible to compile by +//! referencing `__DECLARE_RCC_ATOMIC_ENV`, an identifier it never defines anywhere; the only legal +//! callers are inside `PERIPH_RCC_ATOMIC()`, which takes a FreeRTOS spinlock. There is no FreeRTOS +//! here and core 1 is held in reset at power-on (`hp_sys_clkrst_reg.h`: REG_RST_EN_CORE1_GLOBAL +//! defaults to 1), so masking interrupts around the read-modify-write is sufficient and is what +//! `atomically` does. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +// The four shared registers this file touches. Which field lives in which register is not derivable +// from the macro names - `HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_S` does not say `SOC_CLK_CTRL2` - so the +// pairing is taken from ESP-IDF's own LL, cited per peripheral below. +const soc_clk_ctrl1 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL1_REG); +const soc_clk_ctrl2 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL2_REG); +const soc_clk_ctrl3 = Reg.at(regs.HP_SYS_CLKRST_SOC_CLK_CTRL3_REG); +/// SDMMC's reset bit is not in HP_SYS_CLKRST at all. `sdmmc_ll_reset_register` +/// (`sdmmc_ll.h:158-163`) writes `LP_AON_CLKRST.hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc`, a register +/// of the *low-power* always-on clock-and-reset block, which it shares with the Ethernet MAC. So +/// the `Gates.reset` field is a register as well as a bit, and this is the row that proves it has +/// to be. +const lp_hp_sdmmc_emac_rst_ctrl = Reg.at(regs.LP_CLKRST_HP_SDMMC_EMAC_RST_CTRL_REG); +const hp_rst_en1 = Reg.at(regs.HP_SYS_CLKRST_HP_RST_EN1_REG); + +/// Interrupts masked for the duration of a read-modify-write on a shared register: +/// +/// const guard = clkrst.maskInterrupts(); +/// defer guard.release(); +/// +/// mstatus.MIE is bit 3. `csrrc` clears it and returns the previous mstatus in one instruction, and +/// `release` restores only what was actually there - so this composes: using it inside code that +/// already had interrupts off does not turn them on at the end. +pub const Guard = struct { + prev_mie: bool, + + pub inline fn release(self: Guard) void { + if (self.prev_mie) { + asm volatile ("csrs mstatus, %[mask]" + : + : [mask] "r" (@as(u32, 1 << 3)), + ); + } + } +}; + +pub inline fn maskInterrupts() Guard { + const prev = asm volatile ("csrrc %[out], mstatus, %[mask]" + : [out] "=r" (-> u32), + : [mask] "r" (@as(u32, 1 << 3)), + ); + return .{ .prev_mie = prev & (1 << 3) != 0 }; +} + +/// A peripheral's clock gates and reset bit. +/// +/// `sys_clk` is present only where the peripheral has a second gate on the SYS clock as well as the +/// APB one; UART has both (uart_ll.h:252-253 reads `soc_clk_ctrl2.reg_uart0_apb_clk_en` and +/// `soc_clk_ctrl1.reg_uart0_sys_clk_en`), most blocks have only APB. +const Gates = struct { + apb_clk: ?struct { reg: Reg, field: Field } = null, + sys_clk: ?struct { reg: Reg, field: Field } = null, + reset: struct { reg: Reg, field: Field }, + /// TIMG only: resetting the block re-arms flash-boot protection, which reboots the board a + /// moment later with no diagnostic. `timg_ll.h:53-72` documents it and clears the bit as part of + /// the reset; anything that resets TIMG must do the same. + clears_flashboot: bool = false, +}; + +pub const Peripheral = enum { + uart0, + uart1, + uart2, + uart3, + uart4, + timg0, + timg1, + systimer, + twai0, + ledc, + i2c0, + i2c1, + sdmmc, + + fn gates(comptime self: Peripheral) Gates { + return switch (self) { + // uart_ll.h:251-253 for UART0, and the same three fields per instance after it. + .uart0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART0_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART0_APB_V) }, + }, + .uart1 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART1_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART1_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART1_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART1_APB_V) }, + }, + .uart2 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART2_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART2_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART2_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART2_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART2_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART2_APB_V) }, + }, + .uart3 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART3_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART3_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART3_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART3_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART3_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART3_APB_V) }, + }, + .uart4 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART4_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART4_APB_CLK_EN_V) }, + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_UART4_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART4_SYS_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_UART4_APB_S, regs.HP_SYS_CLKRST_REG_RST_EN_UART4_APB_V) }, + }, + // timg_ll.h:35-42 for the gate, :60-72 for the reset. The timer groups' APB gate is in + // SOC_CLK_CTRL2 - the same word as the UARTs' - not in PERI_CLK_CTRL21. An earlier + // version of this table had these four entries in PERI_CLK_CTRL21 and so wrote bits + // 21-24 of an unrelated register; hp_sys_clkrst_reg.h:605 defines SOC_CLK_CTRL2_REG and + // :753/:763/:770/:777 put TIMERGRP0 at bit 21, TIMERGRP1 at 22, SYSTIMER at 23 and + // TWAI0 at 24 inside it. PERI_CLK_CTRL20/21 do hold timer-group fields - the per-timer + // clock source and gate, see hal/timg.zig - which is what made the mix-up plausible. + // + // It survived a hardware check because `isClockEnabled` read back the same wrong bit + // `setClockEnabled` had just written: self-consistent, and independent of the chip. + .timg0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP0_S, regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP0_V) }, + .clears_flashboot = true, + }, + .timg1 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP1_S, regs.HP_SYS_CLKRST_REG_RST_EN_TIMERGRP1_V) }, + .clears_flashboot = true, + }, + // systimer_ll.h:71-72. + .systimer => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_SYSTIMER_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SYSTIMER_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_STIMER_S, regs.HP_SYS_CLKRST_REG_RST_EN_STIMER_V) }, + }, + // The one block whose clock is gated OFF at power-on, which makes it the only peripheral + // where `enable` is observably necessary rather than merely correct. + .twai0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_TWAI0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TWAI0_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_TWAI0_S, regs.HP_SYS_CLKRST_REG_RST_EN_TWAI0_V) }, + }, + // ledc_ll.h:135 for the gate (`HP_SYS_CLKRST.soc_clk_ctrl3.reg_ledc_apb_clk_en`) and + // :150 for the reset (`hp_rst_en1.reg_rst_en_ledc`). LEDC's APB gate is the *first* bit + // of SOC_CLK_CTRL3, a third register this table did not previously need, and it is one + // of the few whose reset value is 0 (hp_sys_clkrst_reg.h:835): LEDC's registers are + // gated off at power-on, so `setClockEnabled(.ledc, true)` is a prerequisite and not a + // formality. LEDC's *function* clock and its source mux live in PERI_CLK_CTRL22 + // (ledc_ll.h:179, :241) and belong to the peripheral, not to this table - see + // hal/ledc.zig. + .ledc => .{ + .apb_clk = .{ .reg = soc_clk_ctrl3, .field = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_LEDC_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_S, regs.HP_SYS_CLKRST_REG_RST_EN_LEDC_V) }, + }, + // i2c_ll.h:149-156 for the gates (`HP_SYS_CLKRST.soc_clk_ctrl2.reg_i2c0_apb_clk_en`, + // and `reg_i2c1_apb_clk_en` for port 1) and :167-176 for the resets + // (`hp_rst_en1.reg_rst_en_i2c0` / `_i2c1`). Both APB gates default to 1 + // (hp_sys_clkrst_reg.h:694-703), so the registers are reachable from boot; what I2C + // does *not* get from this table is its controller clock, whose enable, source mux and + // divider are I2C-specific fields of PERI_CLK_CTRL10/11 and live in hal/i2c.zig. That + // one defaults to 0, so an I2C port brought up through this table alone has readable + // registers and a state machine that never moves. + .i2c0 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_S, regs.HP_SYS_CLKRST_REG_RST_EN_I2C0_V) }, + }, + .i2c1 => .{ + .apb_clk = .{ .reg = soc_clk_ctrl2, .field = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_APB_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_APB_CLK_EN_V) }, + .reset = .{ .reg = hp_rst_en1, .field = Field.of(regs.HP_SYS_CLKRST_REG_RST_EN_I2C1_S, regs.HP_SYS_CLKRST_REG_RST_EN_I2C1_V) }, + }, + // The one row in this table whose two halves live in two different peripherals, and + // the one whose clock really is gated off at power-on alongside LEDC's. + // + // `sdmmc_ll.h:140-144` is the gate: `HP_SYS_CLKRST.soc_clk_ctrl1.reg_sdmmc_sys_clk_en`, + // a *SYS* clock and not an APB one - SDMMC has no APB gate at all, which is why the + // `apb_clk` field is absent here rather than merely unused. It defaults to 0 + // (hp_sys_clkrst_reg.h:475-481, "default: 0"), so `setClockEnabled(.sdmmc, true)` is a + // prerequisite for the register block reading anything but stale values. + // + // `sdmmc_ll.h:158-163` is the reset, and it is in LP_AON_CLKRST: + // `hp_sdmmc_emac_rst_ctrl.rst_en_sdmmc`, bit 28 (lp_clkrst_reg.h:993-999). Looking for + // an `HP_SYS_CLKRST_REG_RST_EN_SDMMC` finds nothing, which is exactly the shape of the + // mistake the timer-group rows above record: a plausible name in the wrong register. + // + // The host clock generator - source mux, divider, sampling phase - is *not* here. It + // is SDMMC-specific and lives in PERI_CLK_CTRL01/02, in hal/sdmmc.zig, the same + // division this table makes for I2C and LEDC. + .sdmmc => .{ + .sys_clk = .{ .reg = soc_clk_ctrl1, .field = Field.of(regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDMMC_SYS_CLK_EN_V) }, + .reset = .{ .reg = lp_hp_sdmmc_emac_rst_ctrl, .field = Field.of(regs.LP_CLKRST_RST_EN_SDMMC_S, regs.LP_CLKRST_RST_EN_SDMMC_V) }, + }, + }; + } +}; + +/// Turn a peripheral's bus clocks on or off. +pub fn setClockEnabled(comptime p: Peripheral, on: bool) void { + const g = comptime p.gates(); + const v: u32 = @intFromBool(on); + const guard = maskInterrupts(); + defer guard.release(); + if (g.sys_clk) |s| s.reg.modify(.{s.field.is(v)}); + if (g.apb_clk) |a| a.reg.modify(.{a.field.is(v)}); +} + +/// Whether the peripheral's bus clock is on. +/// +/// APB gate if it has one, SYS gate otherwise: SDMMC has only the latter (`sdmmc_ll.h:140-144`), +/// and answering `true` unconditionally for it would have made the oracle's clock check - the one +/// that exists because a gated block reads stale rather than zero - pass on a gated block. +pub fn isClockEnabled(comptime p: Peripheral) bool { + const g = comptime p.gates(); + if (g.apb_clk) |a| return a.reg.get(a.field) == 1; + if (g.sys_clk) |s| return s.reg.get(s.field) == 1; + return true; +} + +/// Pulse a peripheral's reset: assert, deassert. +/// +/// For the timer groups this also clears flash-boot watchdog protection, which the reset re-arms. +/// Leaving that out reboots the board a moment later with nothing on the console to explain it. +pub fn resetPeripheral(comptime p: Peripheral) void { + const g = comptime p.gates(); + { + const guard = maskInterrupts(); + defer guard.release(); + g.reset.reg.modify(.{g.reset.field.is(1)}); + g.reset.reg.modify(.{g.reset.field.is(0)}); + } + if (comptime g.clears_flashboot) { + const wdtconfig0 = Reg.atAddress(switch (p) { + .timg0 => regs.TIMG_WDTCONFIG0_REG(0), + .timg1 => regs.TIMG_WDTCONFIG0_REG(1), + else => unreachable, + }); + wdtconfig0.modify(.{Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V).is(0)}); + } +} + +/// Reset a peripheral and make sure its clocks are on, in that order: a peripheral configured +/// before its reset is released loses the configuration. +pub fn init(comptime p: Peripheral) void { + setClockEnabled(p, true); + resetPeripheral(p); +} diff --git a/src/hal/gpio.zig b/src/hal/gpio.zig new file mode 100644 index 0000000..88a8675 --- /dev/null +++ b/src/hal/gpio.zig @@ -0,0 +1,471 @@ +//! GPIO and the IO MUX. +//! +//! The P4 has 57 pins (GPIO0-56) and every whole-bank register is therefore split in two: `out` +//! covers 0-31 and `out1` covers 32-56. Getting that split wrong is the classic P4 GPIO bug - a +//! write to `out` with a shift of 40 lands on pin 8 - so the bank arithmetic lives in exactly one +//! place here (`Bank`) and every operation goes through it. +//! +//! Levels and enables are driven through the `_W1TS`/`_W1TC` (write-1-to-set / write-1-to-clear) +//! aliases rather than read-modify-write on `out`/`enable`. That is what ESP-IDF's LL does, and it +//! is not a style choice: a read-modify-write of a whole bank races anything else touching another +//! pin in the same bank, and there is no lock here to prevent it. +//! +//! Pad configuration (direction of the *input* buffer, pulls, drive strength, function select) is +//! not in the GPIO peripheral at all - it is in the IO MUX, one register per pad. The two must be +//! kept in step: a pin driven by `enable` but with `fun_ie` clear cannot be read back, which is the +//! single most common "my GPIO does not work" on this part. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// GPIO0-56. 57 pins, and the last five (52-56) exist only on some packages. +pub const max_pin = 56; +pub const pin_count = max_pin + 1; + +/// Which half of a split bank register a pin lives in, and its bit inside that half. +const Bank = struct { + high: bool, + bit: u5, + + inline fn of(pin: u8) Bank { + std.debug.assert(pin <= max_pin); + return if (pin < 32) + .{ .high = false, .bit = @intCast(pin) } + else + .{ .high = true, .bit = @intCast(pin - 32) }; + } + + inline fn mask(self: Bank) u32 { + return @as(u32, 1) << self.bit; + } + + inline fn pick(self: Bank, lo: Reg, hi: Reg) Reg { + return if (self.high) hi else lo; + } +}; + +// The whole-bank registers. `_W1TS`/`_W1TC` are separate addresses that set or clear only the bits +// written as 1, which is what makes a single-pin update atomic against the rest of the bank. +const out = Reg.at(regs.GPIO_OUT_REG); +const out1 = Reg.at(regs.GPIO_OUT1_REG); +const out_w1ts = Reg.at(regs.GPIO_OUT_W1TS_REG); +const out1_w1ts = Reg.at(regs.GPIO_OUT1_W1TS_REG); +const out_w1tc = Reg.at(regs.GPIO_OUT_W1TC_REG); +const out1_w1tc = Reg.at(regs.GPIO_OUT1_W1TC_REG); +const enable_w1ts = Reg.at(regs.GPIO_ENABLE_W1TS_REG); +const enable1_w1ts = Reg.at(regs.GPIO_ENABLE1_W1TS_REG); +const enable_w1tc = Reg.at(regs.GPIO_ENABLE_W1TC_REG); +const enable1_w1tc = Reg.at(regs.GPIO_ENABLE1_W1TC_REG); +const enable = Reg.at(regs.GPIO_ENABLE_REG); +const enable1 = Reg.at(regs.GPIO_ENABLE1_REG); +const in = Reg.at(regs.GPIO_IN_REG); +const in1 = Reg.at(regs.GPIO_IN1_REG); + +/// One IO MUX register per pad, stride taken from two consecutive macros rather than assumed. +const pad = mmio.RegArray( + regs.PERIPHS_IO_MUX_U_PAD_GPIO0, + regs.PERIPHS_IO_MUX_U_PAD_GPIO1, + pin_count, +); + +// Pad fields. These macros are unprefixed globals in io_mux_reg.h - they describe every pad, not +// one - which is why they read as bare `MCU_SEL` rather than `IO_MUX_GPIO7_MCU_SEL`. +const fun_ie = Field.of(regs.FUN_IE_S, regs.FUN_IE_V); +const fun_drv = Field.of(regs.FUN_DRV_S, regs.FUN_DRV_V); +const mcu_sel = Field.of(regs.MCU_SEL_S, regs.MCU_SEL_V); +// io_mux_reg.h defines no macros for the two pull bits; io_mux_struct.h documents them as +// `fun_wpd : R/W; bitpos: [7]` and `fun_wpu : R/W; bitpos: [8]`. +const fun_wpd = Field.bit(7); +const fun_wpu = Field.bit(8); + +/// IO MUX function for a pad. Function 1 is plain GPIO on every P4 pad; the others select a +/// peripheral wired directly to that pad, and anything not on this list has to go through the GPIO +/// matrix instead. +pub const Function = enum(u3) { + f0 = 0, + /// Plain GPIO - the GPIO peripheral drives and samples the pad. + gpio = 1, + f2 = 2, + f3 = 3, + f4 = 4, + f5 = 5, + f6 = 6, + f7 = 7, +}; + +pub const Drive = enum(u2) { + /// ~5 mA + weakest = 0, + /// ~10 mA + weak = 1, + /// ~20 mA, the reset value + medium = 2, + /// ~40 mA + strong = 3, +}; + +pub const Pull = enum { none, up, down }; + +// ------------------------------------------------------------------------------------- levels + +/// Drive a pin high or low. Uses the write-1-to-set/clear alias, so no other pin in the bank is +/// disturbed and no read is needed. +pub inline fn setLevel(pin: u8, level: u1) void { + const b = Bank.of(pin); + const r = if (level == 1) + b.pick(out_w1ts, out1_w1ts) + else + b.pick(out_w1tc, out1_w1tc); + r.writeRaw(b.mask()); +} + +pub inline fn setHigh(pin: u8) void { + setLevel(pin, 1); +} + +pub inline fn setLow(pin: u8) void { + setLevel(pin, 0); +} + +pub inline fn toggle(pin: u8) void { + const b = Bank.of(pin); + if (b.pick(out, out1).raw() & b.mask() != 0) setLow(pin) else setHigh(pin); +} + +/// Sample the pad. Reads the *input* register, so it reports what the pin is actually at - which +/// for an open-drain or externally driven pin is not necessarily what was last written to `out`. +/// Requires the pad's input buffer to be enabled (`setInputEnable`). +pub inline fn getLevel(pin: u8) u1 { + const b = Bank.of(pin); + return @intCast((b.pick(in, in1).raw() >> b.bit) & 1); +} + +/// What was last driven, from the output register rather than the pad. +pub inline fn getDrivenLevel(pin: u8) u1 { + const b = Bank.of(pin); + return @intCast((b.pick(out, out1).raw() >> b.bit) & 1); +} + +// -------------------------------------------------------------------------------- direction + +pub inline fn outputEnable(pin: u8) void { + const b = Bank.of(pin); + b.pick(enable_w1ts, enable1_w1ts).writeRaw(b.mask()); +} + +pub inline fn outputDisable(pin: u8) void { + const b = Bank.of(pin); + b.pick(enable_w1tc, enable1_w1tc).writeRaw(b.mask()); +} + +pub inline fn isOutputEnabled(pin: u8) bool { + const b = Bank.of(pin); + return b.pick(enable, enable1).raw() & b.mask() != 0; +} + +/// The pad's input buffer. Independent of the output driver: both can be on at once, which is how a +/// pin is read back while being driven. +pub inline fn setInputEnable(pin: u8, on: bool) void { + pad.at(pin).modify(.{fun_ie.is(@intFromBool(on))}); +} + +/// Whether the pad's input buffer is on. The counterpart of `setInputEnable`, and worth having +/// because a routed input with `fun_ie` clear is indistinguishable from a card that never drove +/// the pin: both read as a constant. +pub inline fn isInputEnabled(pin: u8) bool { + return pad.at(pin).get(fun_ie) != 0; +} + +// -------------------------------------------------------------------------------- pad config + +pub inline fn setFunction(pin: u8, f: Function) void { + pad.at(pin).modify(.{mcu_sel.is(@intFromEnum(f))}); +} + +pub inline fn setDrive(pin: u8, d: Drive) void { + pad.at(pin).modify(.{fun_drv.is(@intFromEnum(d))}); +} + +/// Internal pull resistors. Setting one direction always clears the other in the same store: a pad +/// with both enabled is a fight between two resistors, and it is easy to reach by two calls. +pub inline fn setPull(pin: u8, p: Pull) void { + pad.at(pin).modify(.{ + fun_wpu.is(@intFromBool(p == .up)), + fun_wpd.is(@intFromBool(p == .down)), + }); +} + +/// What `setPull` last left, read back from the pad. A pad with both resistors enabled cannot be +/// reached through `setPull`, but the reset value or another driver can leave one that way, so the +/// contradictory case is reported as `.none` rather than picking a winner. +pub inline fn getPull(pin: u8) Pull { + const w = pad.at(pin).raw(); + const up = w & fun_wpu.mask() != 0; + const down = w & fun_wpd.mask() != 0; + if (up and !down) return .up; + if (down and !up) return .down; + return .none; +} + +/// Open-drain: the pad drives low and releases high instead of driving both rails. +/// +/// This one is not in the IO MUX with the other pad properties - it is `GPIO_PINn_PAD_DRIVER`, bit +/// 2 of the GPIO peripheral's per-pin register (`gpio_reg.h:363-368`, "1:open-drain. 0:normal"), +/// which is a different register file from `PERIPHS_IO_MUX_U_PAD_GPIOn`. A shared bus - I2C, or any +/// wired-AND signal - needs this on both pads *and* an external pull-up; the internal pull-up is +/// too weak for anything but a short trace at a low bit rate. +pub inline fn setOpenDrain(pin: u8, on: bool) void { + pin_cfg.at(pin).modify(.{pad_driver.is(@intFromBool(on))}); +} + +/// The GPIO peripheral's per-pin configuration register, one per pad. Not the IO MUX: this file +/// holds the open-drain select, the interrupt configuration and the input synchroniser bypasses. +const pin_cfg = mmio.RegArray(regs.GPIO_PIN0_REG, regs.GPIO_PIN1_REG, pin_count); +const pad_driver = Field.of(regs.GPIO_PIN0_PAD_DRIVER_S, regs.GPIO_PIN0_PAD_DRIVER_V); + +// -------------------------------------------------------------------------- pin interrupts + +/// How a pad raises its interrupt. `gpio_reg.h:377-381`: "0:disable GPIO interrupt. 1:trigger at +/// posedge. 2:trigger at negedge. 3:trigger at any edge. 4:valid at low level. 5:valid at high +/// level". +pub const IntrType = enum(u3) { + disable = 0, + posedge = 1, + negedge = 2, + anyedge = 3, + low_level = 4, + high_level = 5, +}; + +const int_type = Field.of(regs.GPIO_PIN0_INT_TYPE_S, regs.GPIO_PIN0_INT_TYPE_V); +/// Five bits, one per consumer of the pad's interrupt, not a boolean. `gpio_reg.h:400-402` says +/// "set bit 13 to enable CPU interrupt, set bit 14 to enable CPU(not shielded) interrupt", and +/// `gpio_ll.h:41,213` names bit 0 of the field `GPIO_LL_INTR0_ENA` and writes exactly that to +/// route a pad to the `gpio_intr0` source. Writing 1 here means "line 0", not "enabled". +const int_ena = Field.of(regs.GPIO_PIN0_INT_ENA_S, regs.GPIO_PIN0_INT_ENA_V); + +/// Which of the P4's four GPIO interrupt outputs a pad drives. Each is a separate entry in the +/// interrupt matrix (`hal.intr.Source.gpio_intr0` .. `gpio_intr3`), and each has its own status +/// register pair. ESP-IDF only ever uses line 0 - `gpio_ll_intr_enable_on_core` hard-codes +/// `GPIO_LL_INTR0_ENA` with a "TODO: IDF-7995" beside it - so line 0 is the tested path. +pub const IntrLine = enum(u3) { + line0 = 0, + line1 = 1, + line2 = 2, + line3 = 3, +}; + +/// Per-line status, gated by `int_ena`. Reading `status`/`status1` instead would report pads whose +/// interrupt is configured but routed to a different line. `gpio_reg.h:277,291` for line 0, +/// `:302,316` for line 1; lines 2 and 3 continue the same +0x8 stride. +const intr_status = mmio.RegArray(regs.GPIO_INTR_0_REG, regs.GPIO_INTR_1_REG, 4); +const intr_status1 = mmio.RegArray(regs.GPIO_INTR1_0_REG, regs.GPIO_INTR1_1_REG, 4); + +/// Status is cleared through a shared write-1-to-clear register, not a per-line one: one pad has +/// one latch however many lines observe it. `gpio_reg.h:233,269`. +const status_w1tc = Reg.at(regs.GPIO_STATUS_W1TC_REG); +const status1_w1tc = Reg.at(regs.GPIO_STATUS1_W1TC_REG); + +/// Arm a pad's interrupt and route it to one of the four GPIO interrupt outputs. +/// +/// This is the GPIO peripheral's half only. The other half is `hal.intr`: the chosen line still +/// has to be routed from `Source.gpio_intr0`+n to a CLIC line and given a handler. Doing it in two +/// calls is deliberate - one pad's interrupt and one CPU line are not the same resource, and +/// several pads normally share a line. +/// +/// Stale latched status is cleared first. A pad that saw an edge before its interrupt was armed +/// otherwise fires immediately on enable, which looks exactly like a real event. +pub fn setInterrupt(pin: u8, t: IntrType, line: IntrLine) void { + std.debug.assert(pin <= max_pin); + clearInterrupt(pin); + pin_cfg.at(pin).modify(.{ + int_type.is(@intFromEnum(t)), + int_ena.is(if (t == .disable) 0 else @as(u32, 1) << @intFromEnum(line)), + }); +} + +/// Disarm, leaving the trigger type alone so it can be re-enabled unchanged. +pub fn disableInterrupt(pin: u8) void { + pin_cfg.at(pin).modify(.{int_ena.is(0)}); +} + +pub fn interruptPending(pin: u8, line: IntrLine) bool { + const b = Bank.of(pin); + const i: u32 = @intFromEnum(line); + return b.pick(intr_status.at(i), intr_status1.at(i)).raw() & b.mask() != 0; +} + +/// Every pad currently interrupting on `line`, as a 57-bit mask in two halves. One read of each +/// register, so a handler can dispatch the whole set without re-reading between pads. +pub fn pendingMask(line: IntrLine) struct { low: u32, high: u32 } { + const i: u32 = @intFromEnum(line); + return .{ .low = intr_status.at(i).raw(), .high = intr_status1.at(i).raw() }; +} + +pub fn clearInterrupt(pin: u8) void { + const b = Bank.of(pin); + b.pick(status_w1tc, status1_w1tc).writeRaw(b.mask()); +} + +pub fn clearInterrupts(low: u32, high: u32) void { + if (low != 0) status_w1tc.writeRaw(low); + if (high != 0) status1_w1tc.writeRaw(high); +} + +/// Everything a pin needs to be a plain push-pull output, in the order the hardware wants: select +/// the pad's function before enabling the driver, so the pin never spends a moment driven by +/// whatever peripheral the IO MUX happened to be pointing at. +pub fn configureOutput(pin: u8, opts: struct { + drive: Drive = .medium, + /// Enable the input buffer too, so the pin can be read back. + readback: bool = false, +}) void { + setFunction(pin, .gpio); + // Point the matrix at the GPIO peripheral: a pad left routed to whatever signal was there + // before is the failure this line prevents. + func_out_sel.at(pin).modify(.{ out_sel.is(matrix_gpio_signal), oen_sel.is(0) }); + pad.at(pin).modify(.{ + fun_drv.is(@intFromEnum(opts.drive)), + fun_ie.is(@intFromBool(opts.readback)), + fun_wpu.is(0), + fun_wpd.is(0), + }); + outputEnable(pin); +} + +/// A plain input: driver off, input buffer on, optional pull. +pub fn configureInput(pin: u8, opts: struct { pull: Pull = .none }) void { + outputDisable(pin); + setFunction(pin, .gpio); + pad.at(pin).modify(.{ + fun_ie.is(1), + fun_wpu.is(@intFromBool(opts.pull == .up)), + fun_wpd.is(@intFromBool(opts.pull == .down)), + }); +} + +// ------------------------------------------------------------------------------- GPIO matrix + +/// The GPIO matrix: 256 peripheral output signals, any of which can be routed to any pad. This is +/// how a UART reaches a pin that has no direct IO MUX function for it. +const func_out_sel = mmio.RegArray( + regs.GPIO_FUNC0_OUT_SEL_CFG_REG, + regs.GPIO_FUNC1_OUT_SEL_CFG_REG, + pin_count, +); +// The input side of the matrix, indexed by *signal* rather than by pad: GPIO_FUNCn_IN_SEL_CFG +// selects which pad feeds peripheral input signal n. That is the opposite indexing from +// `func_out_sel` above, and it is why the two arrays exist separately. +// +// The base is FUNC1's address minus one word, not FUNC1's address. gpio_struct.h:849 declares +// `func_in_sel_cfg[256]` and notes func0 is reserved, so ESP-IDF's register header defines no +// GPIO_FUNC0_IN_SEL_CFG_REG at all - the array starts at +0x158 with a name-less word. Anchoring +// on FUNC1 with a count of 256 is off by one in both directions: `at(n)` would configure signal +// n+1, and `at(255)` would land on GPIO_FUNC0_OUT_SEL_CFG_REG (+0x558) and start driving a pad. +// Bounds checked against the headers: FUNC255_IN_SEL_CFG_REG is +0x554 = 0x158 + 4*255. +const func_in_sel = mmio.RegArray( + regs.GPIO_FUNC1_IN_SEL_CFG_REG - 4, + regs.GPIO_FUNC1_IN_SEL_CFG_REG, + 256, +); + +const out_sel = Field.of(regs.GPIO_FUNC0_OUT_SEL_S, regs.GPIO_FUNC0_OUT_SEL_V); +const oen_sel = Field.of(regs.GPIO_FUNC0_OEN_SEL_S, regs.GPIO_FUNC0_OEN_SEL_V); + +// The input side's three fields. All of GPIO_FUNCn_IN_SEL_CFG's fields share these shifts, so as +// with the pad registers one macro triple describes all 256. +const in_sel = Field.of(regs.GPIO_FUNC1_IN_SEL_S, regs.GPIO_FUNC1_IN_SEL_V); +const in_inv_sel = Field.of(regs.GPIO_FUNC1_IN_INV_SEL_S, regs.GPIO_FUNC1_IN_INV_SEL_V); +/// 1 = take this signal from the GPIO matrix, 0 = from the pad's direct IO MUX function. +const sig_in_sel = Field.of(regs.GPIO_SIG1_IN_SEL_S, regs.GPIO_SIG1_IN_SEL_V); + +/// Writing this value instead of a peripheral signal index means "the GPIO peripheral drives this +/// pad", which is the matrix's way of expressing plain GPIO output. It comes from IDF's own signal +/// map because it is chip-specific: 256 here, 128 on the ESP32-S3. +pub const matrix_gpio_signal: u32 = regs.SIG_GPIO_OUT_IDX; + +/// Route a peripheral output signal to a pad through the matrix, and let that peripheral own the +/// pad's output enable. +/// +/// `OEN_SEL` reads backwards from its name, and the differential test against ESP-IDF's LL is what +/// caught it: 1 means "use GPIO_ENABLE_REG[n] as the output enable", 0 means "use the peripheral's +/// own output enable signal" (gpio_reg.h, GPIO_FUNC0_OEN_SEL). A routed peripheral must have 0 - its +/// OE is part of the signal being routed. The first version of this function set 1 and then set the +/// matching GPIO_ENABLE bit to compensate, which worked by the wrong mechanism and left the pad +/// latently output-enabled: clear OEN_SEL later and the pin would start driving on its own. +pub fn matrixOut(pin: u8, signal: u32) void { + std.debug.assert(pin <= max_pin); + setFunction(pin, .gpio); + func_out_sel.at(pin).modify(.{ out_sel.is(signal), oen_sel.is(0) }); +} + +/// Route a pad to a peripheral *input* signal through the matrix. +/// +/// Indexed by signal, not by pin, which is the opposite of `matrixOut`: one pad may feed any number +/// of input signals, but a signal has exactly one source. The three writes are one word, where +/// gpio_ll.h:613-618 uses three bitfield stores; the resulting word is identical and nothing here +/// depends on the intermediate states, whereas a driver that read the register back between them +/// could observe a signal sourced from the wrong pad. +/// +/// This does not enable the pad's input buffer - `setInputEnable` does, and a routed input with +/// `fun_ie` clear reads as a constant. Callers that want the pad readable must do both. +pub fn matrixIn(pin: u8, signal: u32) void { + std.debug.assert(pin <= max_pin or pin == matrix_const_zero or pin == matrix_const_one); + std.debug.assert(signal < 256); + func_in_sel.at(signal).modify(.{ + in_sel.is(pin), + in_inv_sel.is(0), + sig_in_sel.is(1), + }); +} + +/// Where a peripheral input signal is sourced from. The read side of `matrixIn`, for a diagnostic +/// that has to distinguish "routed to the wrong pad" from "not routed at all" - the two look the +/// same from the peripheral's end. +pub const MatrixIn = struct { + /// A pad index, or `matrix_const_zero`/`matrix_const_one`. Meaningless when `from_matrix` is + /// false: the field keeps its reset value in that case, which can read like a deliberate + /// tie-high and is not one. + pin: u8, + inverted: bool, + /// `sig_in_sel`. False means the matrix is bypassed entirely and the signal comes from the + /// pad's direct IO MUX function - which for a peripheral that has none is undefined. + from_matrix: bool, +}; + +pub fn matrixInSource(signal: u32) MatrixIn { + std.debug.assert(signal < 256); + const w = func_in_sel.at(signal).raw(); + return .{ + .pin = @intCast((w >> in_sel.shift) & in_sel.unshiftedMask()), + .inverted = w & in_inv_sel.mask() != 0, + .from_matrix = w & sig_in_sel.mask() != 0, + }; +} + +/// Two values of `matrixIn`'s `pin` that are not pins: they tie the signal to a constant level +/// inside the matrix. `gpio_reg.h:3717-3719` documents the encoding on the register itself - +/// "s=0-56: connect GPIO[s] to this port. s=0x3F: set this port always high level. s=0x3E: set +/// this port always low level" - and `soc/gpio_pins.h:13-14` gives them the names ESP-IDF's +/// drivers use. They are chip-specific: 0x38/0x30 on the ESP32, 0x1E/0x1F on the C3. +/// +/// This is how an unwired peripheral input gets a defined level. Leaving one alone is not +/// equivalent: `in_sel` does default to 0x3F, but `sig_in_sel` defaults to 0, which bypasses the +/// matrix entirely and takes the signal from the pad's direct IO MUX function - which for a +/// peripheral that has none is not a constant anything. `hal/sdmmc.zig` needs both of these for +/// slot 1's card-detect and card-interrupt inputs. +pub const matrix_const_one: u8 = 0x3f; +pub const matrix_const_zero: u8 = 0x3e; + +test "bank arithmetic splits at 32, which is where the P4's second register begins" { + try std.testing.expectEqual(@as(u5, 20), Bank.of(20).bit); + try std.testing.expect(!Bank.of(20).high); + try std.testing.expectEqual(@as(u5, 0), Bank.of(32).bit); + try std.testing.expect(Bank.of(32).high); + try std.testing.expectEqual(@as(u5, 24), Bank.of(56).bit); + try std.testing.expectEqual(@as(u32, 1) << 24, Bank.of(56).mask()); +} diff --git a/src/hal/i2c.zig b/src/hal/i2c.zig new file mode 100644 index 0000000..2a1d8db --- /dev/null +++ b/src/hal/i2c.zig @@ -0,0 +1,1091 @@ +//! I2C0 and I2C1 in master mode, FIFO access, no interrupts and no DMA. +//! +//! Slave mode, LP_I2C and the RAM (non-FIFO) access path are deliberately absent. +//! +//! Three things about this peripheral are not visible in the register headers, and each one is a +//! way for a port to produce a bus that half-works: +//! +//! **1. The timing is a dozen registers computed from one number.** SCL low, SCL high, SCL +//! wait-high, SDA hold, SDA sample, start hold, restart setup, stop hold, stop setup and the +//! timeout exponent all come from a single `half_cycle` derived from the source clock and the wanted +//! SCL frequency, and several of them are written *minus one* while two deliberately are not. The +//! arithmetic is reproduced from ESP-IDF exactly, with the line numbers, in `Timing.calculate` and +//! `applyTiming` below - including the parts that look like bugs and are not. +//! +//! **2. Nothing takes effect until `CONF_UPGATE` is written.** The timing and control registers feed +//! a synchroniser rather than the state machine directly, so a driver that configures the block and +//! starts a transaction without `commitConfig()` runs on the *previous* configuration. It is a +//! write-to-trigger bit that reads back 0, so nothing about the register state afterwards shows +//! whether it was ever written - which is exactly the kind of bug a state-comparing differential +//! test cannot see, so it is called out here instead. ESP-IDF puts the call in the driver +//! (`esp_driver_i2c/i2c_master.c:96`, `i2c_ll_update` at `i2c_ll.h:137-141`), not in the LL +//! functions that write the timing. +//! +//! **3. The command opcode numbers changed after the original ESP32, and this chip's own register +//! header still documents the old ones.** `i2c_struct.h:1009-1021` and `i2c_reg.h:1174-1186` say +//! "0: RSTART, 1: WRITE, 2: READ, 3: STOP, 4: END". ESP-IDF's P4 LL says RESTART=6, WRITE=1, +//! READ=3, STOP=2 (`i2c_ll.h:55-59`), which is what every post-ESP32 target uses (esp32c3, esp32c6 +//! and esp32p4 agree; only `esp32/include/hal/i2c_ll.h:47-51` has the numbers the P4 header's prose +//! describes). The LL is the version the shipping driver runs on silicon, so it is the one here, and +//! `i2c_ref.c` builds its command words from IDF's own `I2C_LL_CMD_*` macros so that the +//! differential test would catch a wrong constant here rather than agreeing with it. +//! +//! And one hazard for anything that snapshots this block: **reading `I2C_DATA_REG` pops the RX +//! FIFO.** See `Data register` below. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// HP I2C instances. LP_I2C is a third `i2c_dev_t` in ESP-IDF (`SOC_I2C_NUM` is 3, `soc_caps.h:310`) +/// but it lives in the LP domain with its own clock and pad rules, and is out of scope here. +pub const port_count: u8 = 2; + +/// Bytes in each direction. `i2c_ll.h:29` (`I2C_LL_FIFO_LEN`); the RAM behind it is 32 bytes at +/// +0x100 (TX) and +0x180 (RX), reachable directly only in non-FIFO mode. +pub const fifo_len: u8 = 32; + +/// Command slots. **Eight on this chip**, not sixteen: `i2c_ll.h:31` says `I2C_LL_CMD_REG_NUM 8`, +/// `i2c_struct.h:1073` declares `command[8]`, and the register header stops at `I2C_COMD7_REG` +/// (+0x74). ESP-IDF's own `i2c_ll_master_write_cmd_reg` doc comment claims "should be less than 16" +/// (`i2c_ll.h:433`) - that comment is stale, and `i2c_ll_master_is_cmd_done` two hundred lines later +/// says 8 (`i2c_ll.h:1043`). Eight slots is why the driver's long transfers end a chunk with an END +/// opcode and continue: there is no room for a command per byte. +pub const cmd_slots: u8 = 8; + +// ------------------------------------------------------------------------------------ registers +// +// One array per register, indexed by port. The stride is checked against I2C1's own macro rather +// than assumed: `REG_I2C_BASE(i)` is `DR_REG_I2C0_BASE + i * 0x1000` (`soc/esp32p4/include/soc/ +// soc.h:24`), which the linker script agrees with (`esp32p4.peripherals.ld:17-18`, I2C0 = +// 0x500C4000, I2C1 = 0x500C5000). + +fn portArray(comptime macro0: anytype, comptime macro1: anytype) type { + return mmio.RegArray(macro0, macro1, port_count); +} + +const scl_low_period = portArray(regs.I2C_SCL_LOW_PERIOD_REG(0), regs.I2C_SCL_LOW_PERIOD_REG(1)); +const ctr = portArray(regs.I2C_CTR_REG(0), regs.I2C_CTR_REG(1)); +const sr = portArray(regs.I2C_SR_REG(0), regs.I2C_SR_REG(1)); +const to = portArray(regs.I2C_TO_REG(0), regs.I2C_TO_REG(1)); +const fifo_st = portArray(regs.I2C_FIFO_ST_REG(0), regs.I2C_FIFO_ST_REG(1)); +const fifo_conf = portArray(regs.I2C_FIFO_CONF_REG(0), regs.I2C_FIFO_CONF_REG(1)); +const data = portArray(regs.I2C_DATA_REG(0), regs.I2C_DATA_REG(1)); +const int_raw = portArray(regs.I2C_INT_RAW_REG(0), regs.I2C_INT_RAW_REG(1)); +const int_clr = portArray(regs.I2C_INT_CLR_REG(0), regs.I2C_INT_CLR_REG(1)); +const int_ena = portArray(regs.I2C_INT_ENA_REG(0), regs.I2C_INT_ENA_REG(1)); +const sda_hold = portArray(regs.I2C_SDA_HOLD_REG(0), regs.I2C_SDA_HOLD_REG(1)); +const sda_sample = portArray(regs.I2C_SDA_SAMPLE_REG(0), regs.I2C_SDA_SAMPLE_REG(1)); +const scl_high_period = portArray(regs.I2C_SCL_HIGH_PERIOD_REG(0), regs.I2C_SCL_HIGH_PERIOD_REG(1)); +const scl_start_hold = portArray(regs.I2C_SCL_START_HOLD_REG(0), regs.I2C_SCL_START_HOLD_REG(1)); +const scl_rstart_setup = portArray(regs.I2C_SCL_RSTART_SETUP_REG(0), regs.I2C_SCL_RSTART_SETUP_REG(1)); +const scl_stop_hold = portArray(regs.I2C_SCL_STOP_HOLD_REG(0), regs.I2C_SCL_STOP_HOLD_REG(1)); +const scl_stop_setup = portArray(regs.I2C_SCL_STOP_SETUP_REG(0), regs.I2C_SCL_STOP_SETUP_REG(1)); +const filter_cfg = portArray(regs.I2C_FILTER_CFG_REG(0), regs.I2C_FILTER_CFG_REG(1)); +const comd0 = portArray(regs.I2C_COMD0_REG(0), regs.I2C_COMD0_REG(1)); +const scl_sp_conf = portArray(regs.I2C_SCL_SP_CONF_REG(0), regs.I2C_SCL_SP_CONF_REG(1)); + +/// First address of a port's register block, for the differential harness's window. +pub inline fn base(port: u8) u32 { + std.debug.assert(port < port_count); + return @intCast(scl_low_period.base + scl_low_period.stride * port); +} + +// I2C_CTR_REG. `trans_start`, `fsm_rst` and `conf_upgate` are write-to-trigger: they read back 0, +// so a read-modify-write of this register does not re-trigger them. +const sda_force_out = Field.of(regs.I2C_SDA_FORCE_OUT_S, regs.I2C_SDA_FORCE_OUT_V); +const scl_force_out = Field.of(regs.I2C_SCL_FORCE_OUT_S, regs.I2C_SCL_FORCE_OUT_V); +const rx_full_ack_level = Field.of(regs.I2C_RX_FULL_ACK_LEVEL_S, regs.I2C_RX_FULL_ACK_LEVEL_V); +const ms_mode = Field.of(regs.I2C_MS_MODE_S, regs.I2C_MS_MODE_V); +const trans_start = Field.of(regs.I2C_TRANS_START_S, regs.I2C_TRANS_START_V); +const tx_lsb_first = Field.of(regs.I2C_TX_LSB_FIRST_S, regs.I2C_TX_LSB_FIRST_V); +const rx_lsb_first = Field.of(regs.I2C_RX_LSB_FIRST_S, regs.I2C_RX_LSB_FIRST_V); +const arbitration_en = Field.of(regs.I2C_ARBITRATION_EN_S, regs.I2C_ARBITRATION_EN_V); +const fsm_rst = Field.of(regs.I2C_FSM_RST_S, regs.I2C_FSM_RST_V); +const conf_upgate = Field.of(regs.I2C_CONF_UPGATE_S, regs.I2C_CONF_UPGATE_V); + +// I2C_SR_REG, all read-only. +const resp_rec = Field.of(regs.I2C_RESP_REC_S, regs.I2C_RESP_REC_V); +const arb_lost = Field.of(regs.I2C_ARB_LOST_S, regs.I2C_ARB_LOST_V); +const bus_busy = Field.of(regs.I2C_BUS_BUSY_S, regs.I2C_BUS_BUSY_V); +const rxfifo_cnt = Field.of(regs.I2C_RXFIFO_CNT_S, regs.I2C_RXFIFO_CNT_V); +const txfifo_cnt = Field.of(regs.I2C_TXFIFO_CNT_S, regs.I2C_TXFIFO_CNT_V); + +// I2C_TO_REG. `time_out_value` is only five bits wide - the timeout is 2^value source-clock cycles, +// so 31 is the largest legal exponent and the arithmetic below never approaches it. +const time_out_value = Field.of(regs.I2C_TIME_OUT_VALUE_S, regs.I2C_TIME_OUT_VALUE_V); +const time_out_en = Field.of(regs.I2C_TIME_OUT_EN_S, regs.I2C_TIME_OUT_EN_V); + +// I2C_FIFO_CONF_REG. `rx_fifo_rst`/`tx_fifo_rst` are annotated R/W, not self-clearing: they hold +// the FIFO in reset until written back to 0, which is why resetting one is two stores. +const rxfifo_wm_thrhd = Field.of(regs.I2C_RXFIFO_WM_THRHD_S, regs.I2C_RXFIFO_WM_THRHD_V); +const txfifo_wm_thrhd = Field.of(regs.I2C_TXFIFO_WM_THRHD_S, regs.I2C_TXFIFO_WM_THRHD_V); +const nonfifo_en = Field.of(regs.I2C_NONFIFO_EN_S, regs.I2C_NONFIFO_EN_V); +const rx_fifo_rst = Field.of(regs.I2C_RX_FIFO_RST_S, regs.I2C_RX_FIFO_RST_V); +const tx_fifo_rst = Field.of(regs.I2C_TX_FIFO_RST_S, regs.I2C_TX_FIFO_RST_V); +const fifo_prt_en = Field.of(regs.I2C_FIFO_PRT_EN_S, regs.I2C_FIFO_PRT_EN_V); + +// Timing fields. Every period is nine bits ([8:0], max 511) except `scl_wait_high_period`, which is +// seven ([15:9], max 127) and shares its register with `scl_high_period`. +const scl_low_period_f = Field.of(regs.I2C_SCL_LOW_PERIOD_S, regs.I2C_SCL_LOW_PERIOD_V); +const scl_high_period_f = Field.of(regs.I2C_SCL_HIGH_PERIOD_S, regs.I2C_SCL_HIGH_PERIOD_V); +const scl_wait_high_period_f = Field.of(regs.I2C_SCL_WAIT_HIGH_PERIOD_S, regs.I2C_SCL_WAIT_HIGH_PERIOD_V); +const sda_hold_time = Field.of(regs.I2C_SDA_HOLD_TIME_S, regs.I2C_SDA_HOLD_TIME_V); +const sda_sample_time = Field.of(regs.I2C_SDA_SAMPLE_TIME_S, regs.I2C_SDA_SAMPLE_TIME_V); +const scl_start_hold_time = Field.of(regs.I2C_SCL_START_HOLD_TIME_S, regs.I2C_SCL_START_HOLD_TIME_V); +const scl_rstart_setup_time = Field.of(regs.I2C_SCL_RSTART_SETUP_TIME_S, regs.I2C_SCL_RSTART_SETUP_TIME_V); +const scl_stop_hold_time = Field.of(regs.I2C_SCL_STOP_HOLD_TIME_S, regs.I2C_SCL_STOP_HOLD_TIME_V); +const scl_stop_setup_time = Field.of(regs.I2C_SCL_STOP_SETUP_TIME_S, regs.I2C_SCL_STOP_SETUP_TIME_V); + +// I2C_FILTER_CFG_REG. Both thresholds are four bits, both filters default *enabled* with a +// threshold of 0 - which filters nothing - so "disable" and "enable with 0" are different words. +const scl_filter_thres = Field.of(regs.I2C_SCL_FILTER_THRES_S, regs.I2C_SCL_FILTER_THRES_V); +const sda_filter_thres = Field.of(regs.I2C_SDA_FILTER_THRES_S, regs.I2C_SDA_FILTER_THRES_V); +const scl_filter_en = Field.of(regs.I2C_SCL_FILTER_EN_S, regs.I2C_SCL_FILTER_EN_V); +const sda_filter_en = Field.of(regs.I2C_SDA_FILTER_EN_S, regs.I2C_SDA_FILTER_EN_V); + +// I2C_SCL_SP_CONF_REG: the hardware bus-clear generator. +const scl_rst_slv_en = Field.of(regs.I2C_SCL_RST_SLV_EN_S, regs.I2C_SCL_RST_SLV_EN_V); +const scl_rst_slv_num = Field.of(regs.I2C_SCL_RST_SLV_NUM_S, regs.I2C_SCL_RST_SLV_NUM_V); + +/// Data register offset in words, for the harness's `no_read` list. See `Data register` below. +pub const data_word_offset: u32 = (0x1c - 0x00) / 4; + +// ----------------------------------------------------------------------------- clocks and reset +// +// I2C has clock control in two places, and the split is not symmetrical between the two ports: +// +// * the APB bus clock gate and the block reset are in HP_SYS_CLKRST's shared registers, and live +// in `clkrst.zig` with every other peripheral's (`i2c_ll.h:149-176`); +// * the *controller* clock - the one the bus state machine runs on - its source select and its +// divider are I2C-specific fields of HP_SYS_CLKRST_PERI_CLK_CTRL10/11, and are here. +// +// The asymmetry is the trap: I2C1's source select and controller-clock enable are in PERI_CLK_CTRL10 +// beside I2C0's (bits 26 and 27, `i2c_ll.h:851-852` and `i2c_ll.h:944-945`), while I2C1's *divider* +// is in PERI_CLK_CTRL11 (`i2c_ll.h:196-199`). Reading the field names alone would put all of I2C1 +// in ctrl11. + +const peri_clk_ctrl10 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL10_REG); +const peri_clk_ctrl11 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL11_REG); + +const i2c0_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_SRC_SEL_V); +const i2c1_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_SRC_SEL_V); +const i2c0_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_EN_V); +const i2c1_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_EN_V); +const i2c0_div_num = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUM_V); +const i2c0_div_numerator = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_NUMERATOR_V); +const i2c0_div_denominator = Field.of(regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C0_CLK_DIV_DENOMINATOR_V); +const i2c1_div_num = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUM_V); +const i2c1_div_numerator = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_NUMERATOR_V); +const i2c1_div_denominator = Field.of(regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_I2C1_CLK_DIV_DENOMINATOR_V); + +/// Controller clock source. Two choices on this chip (`clk_tree_defs.h:486-494`), and the register +/// field is one bit: 0 = XTAL, 1 = RC_FAST (`i2c_ll.h:848-852`). +pub const Source = enum(u1) { + /// 40 MHz on this board, and the default. Accurate, which for a bus with a specified maximum + /// clock is the whole point. + xtal = 0, + /// The internal RC oscillator, ~20 MHz and temperature-dependent. Usable only because I2C is a + /// clocked bus with no baud-rate agreement to keep. + rc_fast = 1, +}; + +/// XTAL frequency on this board, as the source frequency to hand `Timing.calculate` for +/// `Source.xtal`. Fixed by the crystal, not by the clock tree: 40 MHz. +pub const xtal_hz: u32 = 40_000_000; + +/// Select the controller clock source. A read-modify-write of a register shared with the other +/// port's clock fields, so it takes the interrupt guard. +pub fn setSource(port: u8, src: Source) void { + std.debug.assert(port < port_count); + const v: u32 = @intFromEnum(src); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl10.modify(.{if (port == 0) i2c0_clk_src_sel.is(v) else i2c1_clk_src_sel.is(v)}); +} + +/// The controller clock gate, which is *not* the APB gate in `clkrst.zig`: registers stay readable +/// and writable with this off, and only the bus state machine stops. It defaults to 0 +/// (`hp_sys_clkrst_reg.h`, REG_I2C0_CLK_EN default 0), so unlike most peripherals on this chip I2C +/// genuinely needs this call before it will do anything. `_i2c_hal_init` (`i2c_hal.c:52-58`) is +/// where ESP-IDF makes it. +pub fn setControllerClockEnabled(port: u8, on: bool) void { + std.debug.assert(port < port_count); + const v: u32 = @intFromBool(on); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl10.modify(.{if (port == 0) i2c0_clk_en.is(v) else i2c1_clk_en.is(v)}); +} + +// -------------------------------------------------------------------------------------- timing + +/// Everything the bus timing registers need, in source-clock cycles, as ESP-IDF computes it. +/// +/// The field widths are ESP-IDF's: `i2c_hal_clk_config_t` is nine `uint16_t` +/// (`hal/i2c_types.h:46-56`). That matters at the extremes - a value that would exceed 65535 wraps +/// there too - and it is why this is `u16` rather than `u32`. +pub const Timing = struct { + /// Controller clock divider, as a *count*: the register takes this minus one. + clkm_div: u16, + scl_low: u16, + scl_high: u16, + scl_wait_high: u16, + sda_hold: u16, + sda_sample: u16, + /// Both the start-condition and the stop-condition setup time. + setup: u16, + /// Both the start-condition and the stop-condition hold time. + hold: u16, + /// Timeout *exponent*: the bus times out after 2^tout source-clock cycles. + tout: u16, + + /// Reproduce `i2c_ll_master_cal_bus_clk` (`i2c_ll.h:104-128`) exactly. + /// + /// The whole derivation, because every line of it is load-bearing: + /// + /// clkm_div = source / (bus * 1024) + 1 + /// sclk = source / clkm_div + /// half = sclk / bus / 2 + /// + /// The `+ 1` is not rounding, it is a floor: the period registers are nine bits, so `half` must + /// stay under 512, and dividing the source clock until `sclk <= 1024 * bus` is what guarantees + /// it. At 40 MHz that makes `clkm_div` 1 for every bus frequency above 39 kHz and grows it + /// below - 10 kHz gives `clkm_div` 4, `sclk` 10 MHz, `half` 500 - so the divider is not an + /// optional refinement, it is what makes slow buses representable at all. + /// + /// From `half`, in source-clock cycles: + /// + /// scl_low = half + /// scl_wait_high = half/2 - 2 if bus >= 80 kHz, else half/4 + /// scl_high = half - scl_wait_high + /// sda_hold = half/4 + /// sda_sample = half/2 + /// setup = hold = half + /// tout = 32 - clz(5 * half) + 2 + /// + /// `scl_wait_high` is the part of the high period during which the master waits for the slave to + /// release SCL (clock stretching); `scl_high` is the part it drives. They sum to `half`, so the + /// nominal frequency is the same either way, and IDF's own comment (`i2c_ll.h:112-114`) records + /// why the split changes at 80 kHz: below that, too much wait-high measurably *raises* the + /// frequency on real hardware. + /// + /// The `tout` expression is `log2(5 * half) + 2` written with a count-leading-zeros: a timeout + /// of about 20 half-cycles, i.e. ten bit times, rounded up to the next power of two because the + /// register holds an exponent. IDF writes it as + /// `sizeof(half_cycle) * 8 - __builtin_clz(5 * half_cycle) + 2` with `half_cycle` a `uint32_t`, + /// hence the 32 here. + /// + /// Not reproduced: the `HAL_ASSERT` at `i2c_ll.h:126-127` that + /// `scl_wait_high < sda_sample < scl_high`. It holds for every frequency this can be asked for + /// at 40 MHz (checked from 10 kHz to 1 MHz), and an assert that cannot fire is noise; the + /// ordering it protects is a hardware requirement, not something this code can choose. + pub fn calculate(source_hz: u32, bus_hz: u32) Timing { + std.debug.assert(bus_hz > 0); + std.debug.assert(source_hz / 2 > bus_hz); + + const clkm_div: u32 = source_hz / (bus_hz * 1024) + 1; + const sclk_hz: u32 = source_hz / clkm_div; + const half: u32 = sclk_hz / bus_hz / 2; + + const wait_high: u32 = if (bus_hz >= 80_000) half / 2 - 2 else half / 4; + return .{ + .clkm_div = @truncate(clkm_div), + .scl_low = @truncate(half), + .scl_wait_high = @truncate(wait_high), + .scl_high = @truncate(half - wait_high), + .sda_hold = @truncate(half / 4), + .sda_sample = @truncate(half / 2), + .setup = @truncate(half), + .hold = @truncate(half), + // @clz(0) is 32 in Zig where __builtin_clz(0) is undefined in C, so this differs from + // IDF only for half == 0, which the assert above rules out. + .tout = @truncate(32 - @clz(5 * half) + 2), + }; + } +}; + +/// Write a computed `Timing` to the peripheral's ten timing registers and the controller-clock +/// divider - `i2c_ll_master_set_bus_timing` (`i2c_ll.h:190-220`). +/// +/// **Which values are written minus one and which are not is the substance of this function.** +/// Eight of the ten are `value - 1`, because the hardware counts from zero. `scl_high_period` and +/// `scl_wait_high_period` are written as-is, and that asymmetry is deliberate: IDF's comment +/// (`i2c_ll.h:201-205`) says the Technical Reference Manual asks for minus one on those two as well, +/// and that following it measurably produces an SCL a little *faster* than asked for, so they do not +/// subtract. A port that "fixes" this by making all ten consistent gets a bus that is out of spec at +/// the top end and passes every test that does not include an oscilloscope. +/// +/// Subtractions are done in `u32` with wrapping and truncated by the field write, which is what the +/// C does for a `uint16_t` of 0 as well - it is unreachable here anyway, since `calculate` asserts +/// `half >= 1`. +pub fn applyTiming(port: u8, t: Timing) void { + std.debug.assert(port < port_count); + setClockDivider(port, t.clkm_div); + + scl_low_period.at(port).modify(.{scl_low_period_f.is(@as(u32, t.scl_low) -% 1)}); + // One store where IDF does two read-modify-writes of the same register (`i2c_ll.h:207-208`). + // Same final word; a write-trace comparison sees the difference, a state comparison does not. + scl_high_period.at(port).modify(.{ + scl_high_period_f.is(t.scl_high), + scl_wait_high_period_f.is(t.scl_wait_high), + }); + sda_hold.at(port).modify(.{sda_hold_time.is(@as(u32, t.sda_hold) -% 1)}); + sda_sample.at(port).modify(.{sda_sample_time.is(@as(u32, t.sda_sample) -% 1)}); + scl_rstart_setup.at(port).modify(.{scl_rstart_setup_time.is(@as(u32, t.setup) -% 1)}); + scl_stop_setup.at(port).modify(.{scl_stop_setup_time.is(@as(u32, t.setup) -% 1)}); + scl_start_hold.at(port).modify(.{scl_start_hold_time.is(@as(u32, t.hold) -% 1)}); + scl_stop_hold.at(port).modify(.{scl_stop_hold_time.is(@as(u32, t.hold) -% 1)}); + to.at(port).modify(.{ time_out_value.is(t.tout), time_out_en.is(1) }); +} + +/// Compute and apply the timing for a target SCL frequency. The whole point of the file. +/// +/// Does **not** commit: call `commitConfig` when the rest of the configuration is in place. That is +/// ESP-IDF's division too - `_i2c_hal_set_bus_timing` (`i2c_hal.c:27-32`) is calculate-then-write, +/// and the driver commits separately. +pub fn setBusTiming(port: u8, source_hz: u32, bus_hz: u32) void { + applyTiming(port, Timing.calculate(source_hz, bus_hz)); +} + +/// The controller clock divider: register field is the divider *minus one*, with the fractional +/// numerator and denominator zeroed because ESP-IDF does not use them +/// (`i2c_ll.h:193-199`, `i2c_ll.h:229-239`). +pub fn setClockDivider(port: u8, clkm_div: u16) void { + std.debug.assert(port < port_count); + const num: u32 = @as(u32, clkm_div) -% 1; + const guard = clkrst.maskInterrupts(); + defer guard.release(); + if (port == 0) { + peri_clk_ctrl10.modify(.{ + i2c0_div_num.is(num), + i2c0_div_numerator.is(0), + i2c0_div_denominator.is(0), + }); + } else { + peri_clk_ctrl11.modify(.{ + i2c1_div_num.is(num), + i2c1_div_numerator.is(0), + i2c1_div_denominator.is(0), + }); + } +} + +// The three narrow timing setters, for tuning one condition without recomputing the whole set - a +// slow slave that needs a longer SDA hold, say. +// +// **These do not use the same convention as `applyTiming`, and that is ESP-IDF's inconsistency, not +// a transcription error.** `i2c_ll_master_set_start_timing` writes `scl_rstart_setup = setup` but +// `scl_start_hold = hold - 1` (`i2c_ll.h:452-456`); `i2c_ll_master_set_stop_timing` writes both as +// given (`i2c_ll.h:467-471`); `i2c_ll_set_sda_timing` writes both as given (`i2c_ll.h:482-486`). +// `i2c_ll_master_set_bus_timing`, meanwhile, subtracts one from all six of those +// (`i2c_ll.h:210-217`). The reconciliation is that `cal_bus_clk` produces *cycle counts* and these +// setters take *register values*, with the single exception of `start_hold` - and IDF's own getters +// agree: `i2c_ll_get_start_timing` adds one back to the hold and not to the setup +// (`i2c_ll.h:644-648`), while `i2c_ll_get_stop_timing` adds nothing (`i2c_ll.h:659-663`). Anything +// tidier here would be a different peripheral configuration from the one IDF produces. + +pub fn setStartTiming(port: u8, setup: u32, hold: u32) void { + std.debug.assert(port < port_count); + scl_rstart_setup.at(port).modify(.{scl_rstart_setup_time.is(setup)}); + scl_start_hold.at(port).modify(.{scl_start_hold_time.is(hold -% 1)}); +} + +pub fn setStopTiming(port: u8, setup: u32, hold: u32) void { + std.debug.assert(port < port_count); + scl_stop_setup.at(port).modify(.{scl_stop_setup_time.is(setup)}); + scl_stop_hold.at(port).modify(.{scl_stop_hold_time.is(hold)}); +} + +pub fn setSdaTiming(port: u8, sample: u32, hold: u32) void { + std.debug.assert(port < port_count); + sda_hold.at(port).modify(.{sda_hold_time.is(hold)}); + sda_sample.at(port).modify(.{sda_sample_time.is(sample)}); +} + +/// Timeout exponent for a wanted timeout in microseconds - +/// `i2c_ll_calculate_timeout_us_to_reg_val` (`i2c_ll.h:1060-1065`). +/// +/// `32 - clz(cycles_per_us * timeout_us)` is `log2` rounded *up*, which is the only sensible +/// direction for a bus timeout. IDF's own default for the SCL timeout is 2000 us +/// (`i2c_ll.h:88`). +pub fn timeoutExponent(source_hz: u32, timeout_us: u32) u32 { + const cycles_per_us = source_hz / 1_000_000; + return 32 - @clz(cycles_per_us * timeout_us); +} + +/// Set just the timeout exponent, leaving the enable bit alone - `i2c_ll_set_tout` +/// (`i2c_ll.h:358-361`). The field is five bits: 2^31 source cycles is the longest expressible +/// timeout, which at 40 MHz is 54 seconds. +pub fn setTimeout(port: u8, exponent: u32) void { + std.debug.assert(port < port_count); + to.at(port).modify(.{time_out_value.is(exponent)}); +} + +pub fn setTimeoutEnabled(port: u8, on: bool) void { + std.debug.assert(port < port_count); + to.at(port).modify(.{time_out_en.is(@intFromBool(on))}); +} + +/// Glitch filter: pulses shorter than `cycles` source-clock cycles are ignored on both SDA and SCL. +/// `cycles == 0` disables both filters - `i2c_ll_master_set_filter` (`i2c_ll.h:753-764`). +/// +/// Note what "disable" means here: the two enable bits default to 1 with thresholds of 0, so the +/// reset state is "filtering enabled, filtering nothing", and disabling is not the same word as +/// enabling with a threshold of 0. Passing 0 therefore leaves the thresholds untouched, exactly as +/// IDF does, rather than zeroing them - a difference the register comparison would catch. +pub fn setFilter(port: u8, cycles: u4) void { + std.debug.assert(port < port_count); + const r = filter_cfg.at(port); + if (cycles > 0) { + r.modify(.{ + scl_filter_thres.is(cycles), + sda_filter_thres.is(cycles), + scl_filter_en.is(1), + sda_filter_en.is(1), + }); + } else { + r.modify(.{ scl_filter_en.is(0), sda_filter_en.is(0) }); + } +} + +// ----------------------------------------------------------------------------------- bring-up + +/// Put a port into master mode with the defaults ESP-IDF's `i2c_hal_master_init` establishes +/// (`i2c_hal.c:39-50`), in the same order. +/// +/// The four control bits are one store where IDF does five separate read-modify-writes of the same +/// register; the resulting word is identical. Each one matters: +/// +/// * `ms_mode = 1` - master. +/// * `sda_force_out = scl_force_out = 0` - open drain. The names are inverted: +/// `i2c_ll_enable_pins_open_drain` writes `!enable_od` (`i2c_ll.h:971-975`), so *zero* is +/// open-drain and one is push-pull. Push-pull on a shared bus is a short circuit the moment two +/// devices disagree, so this is the bit that must not be got backwards. +/// * `arbitration_en = 0` - IDF's master init disables arbitration, which defaults to 1. With a +/// single master there is nothing to arbitrate, and a false arbitration-lost abort on a noisy +/// line is worse than none. +/// * `rx_full_ack_level = 0` - ACK, not NACK, when the RX FIFO hits its threshold. +/// * `tx_lsb_first = rx_lsb_first = 0` - MSB first, which is what I2C is. +/// +/// Then both FIFOs are reset, as IDF does, so the block starts with empty FIFOs whatever the +/// previous user left behind. +pub fn initMaster(port: u8) void { + std.debug.assert(port < port_count); + ctr.at(port).modify(.{ + ms_mode.is(1), + sda_force_out.is(0), + scl_force_out.is(0), + arbitration_en.is(0), + rx_full_ack_level.is(0), + tx_lsb_first.is(0), + rx_lsb_first.is(0), + }); + resetTxFifo(port); + resetRxFifo(port); +} + +/// Latch the configuration into the state machine. Write-to-trigger, self-clearing, and required: +/// see note 2 in this file's header. `i2c_ll_update` (`i2c_ll.h:137-141`). +pub inline fn commitConfig(port: u8) void { + ctr.at(port).modify(.{conf_upgate.is(1)}); +} + +/// Reset the master state machine without touching its configuration. Self-clearing in hardware - +/// IDF writes 1 and never writes 0 (`i2c_ll.h:785-789`, "fsm_rst is a self cleared bit"). For a +/// master that has hung mid-transaction; the bus itself may still need `clearBus`. +pub inline fn resetFsm(port: u8) void { + ctr.at(port).modify(.{fsm_rst.is(1)}); +} + +/// Drive up to `pulses` SCL clocks to free a slave that is holding SDA low, then a STOP - +/// `i2c_ll_master_clr_bus` (`i2c_ll.h:803-810`). Nine pulses is IDF's default +/// (`I2C_LL_RESET_SLV_SCL_PULSE_NUM_DEFAULT`, `i2c_ll.h:87`): enough for any slave to finish the +/// byte it is stuck in and see a NACK. +/// +/// The enable bit is cleared *by hardware* when the pulses have been sent, so completion is polled +/// through `isBusClearDone`, and `commitConfig` is needed both to start it and, per IDF's comment, +/// to resynchronise afterwards. Only meaningful with SCL and SDA actually routed to pads. +pub fn clearBus(port: u8, pulses: u5) void { + std.debug.assert(port < port_count); + scl_sp_conf.at(port).modify(.{ scl_rst_slv_num.is(pulses), scl_rst_slv_en.is(1) }); + commitConfig(port); +} + +pub inline fn isBusClearDone(port: u8) bool { + return scl_sp_conf.at(port).get(scl_rst_slv_en) == 0; +} + +/// Open-drain or push-pull SCL and SDA, at the peripheral end. +/// +/// **The register fields are the inverse of this argument.** `i2c_ll_enable_pins_open_drain` writes +/// `sda_force_out = scl_force_out = !enable_od` (`i2c_ll.h:971-975`), so a zero in either field is +/// what makes that line release instead of driving high. `initMaster` already establishes +/// open-drain; this exists to be able to change it, and to have the polarity checked against IDF's +/// on its own rather than only as part of a seven-field store. +/// +/// This is the *peripheral's* driver behaviour. The pad also has an open-drain bit of its own in the +/// GPIO block (`gpio.setOpenDrain`), and a real bus needs both: the pad hardware must not drive +/// high, and the peripheral must not ask it to. +pub fn setPinsOpenDrain(port: u8, open_drain: bool) void { + std.debug.assert(port < port_count); + const v: u32 = @intFromBool(!open_drain); + ctr.at(port).modify(.{ sda_force_out.is(v), scl_force_out.is(v) }); +} + +// --------------------------------------------------------------------------------------- FIFOs + +/// FIFO or RAM access. FIFO mode is `nonfifo_en = 0`, i.e. the field is the inverse of the name of +/// this function - `i2c_ll_enable_fifo_mode` (`i2c_ll.h:345-348`). +pub fn setFifoMode(port: u8, fifo: bool) void { + std.debug.assert(port < port_count); + fifo_conf.at(port).modify(.{nonfifo_en.is(@intFromBool(!fifo))}); +} + +/// Hold the TX FIFO in reset, then release it. Two stores, because the bit is plain R/W and not +/// self-clearing: writing only the 1 leaves the FIFO permanently reset and every subsequent +/// transmission silently empty (`i2c_ll.h:248-253`). +pub fn resetTxFifo(port: u8) void { + std.debug.assert(port < port_count); + const r = fifo_conf.at(port); + r.modify(.{tx_fifo_rst.is(1)}); + r.modify(.{tx_fifo_rst.is(0)}); +} + +pub fn resetRxFifo(port: u8) void { + std.debug.assert(port < port_count); + const r = fifo_conf.at(port); + r.modify(.{rx_fifo_rst.is(1)}); + r.modify(.{rx_fifo_rst.is(0)}); +} + +/// FIFO watermark thresholds, and the two side effects ESP-IDF attaches to setting them. +/// +/// `fifo_prt_en` gates the watermark interrupts *and* the overflow/underflow protection +/// (`i2c_reg.h:449-459`), and IDF sets it in both threshold setters +/// (`i2c_ll.h:496-500` and `i2c_ll.h:510-515`), so it is set here rather than left to the caller. +/// +/// The other side effect is less obvious and is copied deliberately: IDF's +/// `i2c_ll_set_rxfifo_full_thr` also writes `ctr.rx_full_ack_level = 0`, in a different register. +/// That is coherent rather than sloppy - an RX threshold means "ACK up to here", and a master that +/// NACKed at the threshold would end the transfer instead of pausing it - but it means this +/// operation touches two registers, and after a peripheral reset (where `rx_full_ack_level` defaults +/// to 1) leaving it out is an observable difference rather than a stylistic one. +pub fn setFifoThresholds(port: u8, tx_empty: u5, rx_full: u5) void { + std.debug.assert(port < port_count); + fifo_conf.at(port).modify(.{ + fifo_prt_en.is(1), + txfifo_wm_thrhd.is(tx_empty), + rxfifo_wm_thrhd.is(rx_full), + }); + ctr.at(port).modify(.{rx_full_ack_level.is(0)}); +} + +// ------------------------------------------------------------------------------- Data register +// +// **Reading `I2C_DATA_REG` pops the RX FIFO.** The register header does not say so - it annotates +// the single field `I2C_FIFO_RDATA` as `HRO` and describes the register as "Rx FIFO read data" +// (`i2c_reg.h:464-474`) - but ESP-IDF's LL settles it: `i2c_ll_read_rxfifo` reads *the same address* +// `len` times into successive bytes of a buffer (`i2c_ll.h:691-697`), which can only produce +// distinct bytes if each read advances the FIFO. The write direction is the same address for the +// other FIFO: `i2c_ll_write_txfifo` stores `len` bytes to `hw->data.val` (`i2c_ll.h:674-680`). One +// address, two FIFOs, both with side effects - the same shape as `UART_FIFO_REG`, and the reason +// this offset is in the differential harness's `no_read` list. + +/// Push bytes into the TX FIFO. In FIFO mode each store is one byte into the FIFO regardless of the +/// width of the access; the FIFO is `fifo_len` deep and there is no flow control here, so the caller +/// must not exceed `txSpace`. +pub fn writeTxFifo(port: u8, bytes: []const u8) void { + std.debug.assert(port < port_count); + std.debug.assert(bytes.len <= fifo_len); + const r = data.at(port); + for (bytes) |b| r.writeRaw(b); +} + +/// Pop bytes out of the RX FIFO. Destructive by construction - see above. +pub fn readRxFifo(port: u8, out: []u8) void { + std.debug.assert(port < port_count); + const r = data.at(port); + for (out) |*b| b.* = @truncate(r.raw()); +} + +/// Bytes waiting in the RX FIFO. +pub inline fn rxCount(port: u8) u32 { + return sr.at(port).get(rxfifo_cnt); +} + +/// Bytes queued in the TX FIFO. +pub inline fn txCount(port: u8) u32 { + return sr.at(port).get(txfifo_cnt); +} + +/// Room left in the TX FIFO, saturating at 0 the way `i2c_ll_get_txfifo_len` does +/// (`i2c_ll.h:604-608`) - the counter can read `fifo_len` and the subtraction must not wrap. +pub inline fn txSpace(port: u8) u32 { + const used = txCount(port); + return if (used >= fifo_len) 0 else fifo_len - used; +} + +pub inline fn isBusBusy(port: u8) bool { + return sr.at(port).get(bus_busy) == 1; +} + +// -------------------------------------------------------------------------------- command list +// +// A transaction is up to eight commands written into I2C_COMD0..7 and then triggered as a unit. The +// register header exposes each slot as a single 14-bit field `I2C_COMMANDn` plus a `_DONE` bit at 31 +// and stops there: the sub-fields exist only in `i2c_ll_hw_cmd_t` (`i2c_ll.h:41-52`). So this is one +// of the few places where the field geometry cannot come from a macro pair, and the comptime check +// below is what keeps that honest - the five sub-fields must tile exactly the bits the header calls +// I2C_COMMANDn. + +const cmd_byte_num = Field.of(0, 0xff); +const cmd_ack_en = Field.bit(8); +const cmd_ack_exp = Field.bit(9); +const cmd_ack_val = Field.bit(10); +const cmd_op_code = Field.of(11, 0x7); +const cmd_done = Field.of(regs.I2C_COMMAND0_DONE_S, regs.I2C_COMMAND0_DONE_V); + +comptime { + const command_field = Field.of(regs.I2C_COMMAND0_S, regs.I2C_COMMAND0_V); + const tiled = cmd_byte_num.mask() | cmd_ack_en.mask() | cmd_ack_exp.mask() | + cmd_ack_val.mask() | cmd_op_code.mask(); + if (tiled != command_field.mask()) @compileError( + "the command sub-fields from i2c_ll.h do not tile I2C_COMMAND0 - one of the two headers moved", + ); + if (cmd_done.mask() & command_field.mask() != 0) @compileError("command done bit overlaps the command"); +} + +/// Opcodes, from `i2c_ll.h:55-59`. **Not** the numbers this chip's own register header describes - +/// see note 3 in the file header. +pub const Op = enum(u3) { + write = 1, + stop = 2, + read = 3, + /// Hand the command list back to software with the bus still held, so the next chunk can be + /// loaded. This is how a transfer longer than eight commands or 32 bytes is done without DMA. + end = 4, + /// START, and equally a repeated START. + restart = 6, +}; + +/// One command slot as a value rather than a raw word. +/// +/// The three ACK fields only mean something for one direction each, which is why they are separate +/// rather than one "ack" number: +/// +/// * `ack_check` (WRITE) - compare the ACK bit the slave returns against `ack_expected` and abort +/// the list if it differs. This is what turns a missing device into a NACK error instead of a +/// transfer into the void. +/// * `ack_value` (READ) - the ACK bit this master sends after each byte it reads. Zero (ACK) for +/// every byte but the last, one (NACK) for the last, which is how a slave is told to stop +/// driving the bus. +pub const Command = struct { + op: Op, + /// Bytes to move. Only WRITE and READ use it; a READ of n bytes is one command, not n. + bytes: u8 = 0, + ack_check: bool = false, + ack_expected: u1 = 0, + ack_value: u1 = 0, + + pub inline fn encode(self: Command) u32 { + return (@as(u32, self.bytes) << cmd_byte_num.shift) | + (@as(u32, @intFromBool(self.ack_check)) << cmd_ack_en.shift) | + (@as(u32, self.ack_expected) << cmd_ack_exp.shift) | + (@as(u32, self.ack_value) << cmd_ack_val.shift) | + (@as(u32, @intFromEnum(self.op)) << cmd_op_code.shift); + } +}; + +/// One command slot. The slot stride is checked against the header's own COMD1 macro rather than +/// assumed to be 4. +inline fn cmdReg(port: u8, slot: u8) Reg { + std.debug.assert(slot < cmd_slots); + const stride = comptime mmio.addr(regs.I2C_COMD1_REG(0)) - mmio.addr(regs.I2C_COMD0_REG(0)); + comptime { + // ... and the array is contiguous all the way to the last slot. + if (mmio.addr(regs.I2C_COMD7_REG(0)) != mmio.addr(regs.I2C_COMD0_REG(0)) + stride * 7) + @compileError("the command registers are not a contiguous array of 8"); + } + return Reg.atAddress(comd0.at(port).address + stride * slot); +} + +/// Write a command into a slot. A whole-word store, as IDF's `i2c_ll_master_write_cmd_reg` does +/// (`i2c_ll.h:437-441`): it is the one register here where establishing the entire word is right, +/// because the `done` bit must go back to 0 for the slot to be waited on again. +pub fn writeCommand(port: u8, slot: u8, cmd: Command) void { + std.debug.assert(port < port_count); + cmdReg(port, slot).writeRaw(cmd.encode()); +} + +/// Load a whole command list, in order. Any slot the list does not reach keeps whatever it held - +/// which is harmless, because the sequencer stops at the STOP or END that the list must contain. +pub fn writeCommands(port: u8, cmds: []const Command) void { + std.debug.assert(cmds.len <= cmd_slots); + for (cmds, 0..) |c, i| writeCommand(port, @intCast(i), c); +} + +/// Whether the sequencer has finished a slot. Set by hardware (`R/W/SS`), cleared by writing the +/// slot again. `i2c_ll_master_is_cmd_done` (`i2c_ll.h:1047-1051`). +pub inline fn isCommandDone(port: u8, slot: u8) bool { + return cmdReg(port, slot).get(cmd_done) == 1; +} + +// --------------------------------------------------------------------------------- transactions + +// The master event bits, in I2C_INT_RAW/I2C_INT_ST/I2C_INT_CLR - the same bit numbers in all three +// (`i2c_ll.h:61-70`). Reading INT_RAW is safe: the bits are `R/SS/WTC`, set by hardware and cleared +// only by writing a 1 to the same position in INT_CLR, so polling does not consume them. Writing +// INT_CLR is the one place in this file that must be `writeRaw` rather than `modify`. +const int_trans_complete = Field.of(regs.I2C_TRANS_COMPLETE_INT_RAW_S, regs.I2C_TRANS_COMPLETE_INT_RAW_V); +const int_end_detect = Field.of(regs.I2C_END_DETECT_INT_RAW_S, regs.I2C_END_DETECT_INT_RAW_V); +const int_nack = Field.of(regs.I2C_NACK_INT_RAW_S, regs.I2C_NACK_INT_RAW_V); +const int_arbitration_lost = Field.of(regs.I2C_ARBITRATION_LOST_INT_RAW_S, regs.I2C_ARBITRATION_LOST_INT_RAW_V); +const int_time_out = Field.of(regs.I2C_TIME_OUT_INT_RAW_S, regs.I2C_TIME_OUT_INT_RAW_V); +const int_scl_st_to = Field.of(regs.I2C_SCL_ST_TO_INT_RAW_S, regs.I2C_SCL_ST_TO_INT_RAW_V); +const int_scl_main_st_to = Field.of(regs.I2C_SCL_MAIN_ST_TO_INT_RAW_S, regs.I2C_SCL_MAIN_ST_TO_INT_RAW_V); + +/// The mask ESP-IDF uses for "all interrupts" - `I2C_LL_INTR_MASK`, `i2c_ll.h:1097`. +/// +/// It is 14 bits, and this block has 19 (`I2C_SLAVE_ADDR_UNMATCH_INT` is bit 18). The five it leaves +/// out are slave-mode and general-call events, which is presumably why IDF's mask stops where it +/// does; the value is IDF's rather than a recount so that clearing "everything" means the same thing +/// on both sides of the differential. +pub const all_interrupts: u32 = 0x3fff; + +/// Clear interrupt flags. Write-1-to-clear, so this is a raw store of a mask and never a +/// read-modify-write: reading INT_RAW and writing it back would clear whatever had arrived in +/// between and nothing else. +pub inline fn clearInterrupts(port: u8, mask: u32) void { + int_clr.at(port).writeRaw(mask); +} + +/// Mask every interrupt at the peripheral. This HAL polls; nothing here reaches the CLIC. +/// +/// A whole-word zero rather than IDF's `int_ena &= ~mask` (`i2c_ll.h:305-309`), so it also covers +/// the five slave-mode bits outside `all_interrupts`. Reaching the same word from a block whose +/// `int_ena` reset value is 0 either way, which is why the differential case for it agrees. +pub inline fn disableInterrupts(port: u8) void { + int_ena.at(port).writeRaw(0); +} + +/// How a triggered command list ended. +pub const Outcome = enum { + /// The list ran to its STOP. + complete, + /// The list hit an END opcode: the bus is still held and the next chunk can be loaded. + end_detect, + /// A slave did not acknowledge. The usual meaning is "nothing at that address". + nack, + /// Another master won the bus. Only possible with `arbitration_en` set, which `initMaster` + /// clears. + arbitration_lost, + /// SCL was held low past the configured timeout - `I2C_TO_REG`. Almost always a slave holding + /// the clock, or no pull-up on the line at all. + timeout, + /// The SCL state machine stalled: `scl_st_to` or `scl_main_st_to`. IDF's driver treats this as + /// the signal that a bus deadlock may have happened and `clearBus` is worth trying + /// (`i2c_ll.h:795`). + stalled, + /// Nothing had happened yet. + pending, +}; + +/// Trigger the loaded command list. Write-to-trigger; the bit reads back 0, so this leaves no trace +/// in a register snapshot. `i2c_ll_start_trans` (`i2c_ll.h:629-633`). +pub inline fn startTransaction(port: u8) void { + ctr.at(port).modify(.{trans_start.is(1)}); +} + +/// Read the outcome so far from one load of INT_RAW. +/// +/// Errors are reported ahead of completion, and in the order they matter: an arbitration loss or a +/// NACK can be raised in the same word as `trans_complete`, and calling that transaction complete +/// is how a driver comes to believe a device answered when it did not. +pub fn outcome(port: u8) Outcome { + const raw = int_raw.at(port).raw(); + if (raw & int_arbitration_lost.mask() != 0) return .arbitration_lost; + if (raw & int_nack.mask() != 0) return .nack; + if (raw & int_time_out.mask() != 0) return .timeout; + if (raw & (int_scl_st_to.mask() | int_scl_main_st_to.mask()) != 0) return .stalled; + if (raw & int_trans_complete.mask() != 0) return .complete; + if (raw & int_end_detect.mask() != 0) return .end_detect; + return .pending; +} + +/// Spin until the transaction resolves. Returns `.pending` if it never does, rather than hanging: +/// a bus with no pull-up produces exactly that, and it is a fault to report rather than a board to +/// power-cycle. +/// +/// `spins` is a loop count, not a time. At the ~90 MHz this board boots at, a 100 kHz transfer of a +/// few bytes needs on the order of 10^4 iterations of this loop; the default of 200,000 leaves an +/// order of magnitude of headroom and still returns in well under a second. +pub fn waitTransaction(port: u8, spins: u32) Outcome { + var n: u32 = 0; + while (n < spins) : (n += 1) { + const o = outcome(port); + if (o != .pending) return o; + } + return .pending; +} + +/// The status register's own error bits, which are not the interrupt flags: `resp_rec` is the last +/// ACK level *received* and `arb_lost` is the state machine's own latch. Both are read-only and +/// survive an interrupt clear, so they are what to look at when diagnosing a transfer after the fact. +pub const Status = struct { + /// The ACK bit the slave last returned: 0 = ACK, 1 = NACK. + last_ack: u1, + arbitration_lost: bool, + bus_busy: bool, + rx_bytes: u32, + tx_bytes: u32, +}; + +pub fn status(port: u8) Status { + const raw = sr.at(port).raw(); + return .{ + .last_ack = @intCast((raw >> resp_rec.shift) & 1), + .arbitration_lost = raw & arb_lost.mask() != 0, + .bus_busy = raw & bus_busy.mask() != 0, + .rx_bytes = (raw >> rxfifo_cnt.shift) & rxfifo_cnt.unshiftedMask(), + .tx_bytes = (raw >> txfifo_cnt.shift) & txfifo_cnt.unshiftedMask(), + }; +} + +// ------------------------------------------------------------------------------------ the pads +// +// I2C is a two-wire open-drain bus and the P4 reaches it only through the GPIO matrix: there is no +// IO MUX function for I2C on any pad, so both signals go out through `matrixOut` and come back in +// through `matrixIn`. Both directions are needed even for a write-only master - the master samples +// SDA to read the slave's ACK, and samples SCL to detect stretching - which is why every pad here +// gets its input buffer enabled as well as its driver. + +/// The GPIO matrix signal indices for a port, from ESP-IDF's own signal map +/// (`gpio_sig_map.h:141-148`) via `i2c_periph.c`. On this chip a signal's input and output index +/// happen to be the same number, which is not true on every part and is not something to rely on. +pub fn sclSignal(port: u8) u32 { + return switch (port) { + 0 => regs.I2C0_SCL_PAD_OUT_IDX, + else => regs.I2C1_SCL_PAD_OUT_IDX, + }; +} + +pub fn sdaSignal(port: u8) u32 { + return switch (port) { + 0 => regs.I2C0_SDA_PAD_OUT_IDX, + else => regs.I2C1_SDA_PAD_OUT_IDX, + }; +} + +/// Route SCL and SDA to two pads, open-drain, following `i2c_common_set_pins` +/// (`esp_driver_i2c/i2c_common.c:318-345`) step for step. +/// +/// **The internal pull-ups are not enough for a real bus.** They are on the order of 45 kOhm, which +/// with a few tens of picofarads of trace and device capacitance gives a rise time far past the +/// 1 us that 100 kHz I2C allows. ESP-IDF says the same thing in its own driver documentation and +/// enables them anyway as a convenience for a single device on a short wire. A bus that is expected +/// to work needs external resistors - 4.7 kOhm to 3.3 V is the usual choice at 100 kHz, 2.2 kOhm at +/// 400 kHz - and then `internal_pullups` should be false, because two resistors in parallel is not +/// what either calculation assumed. +/// +/// The order matters in one place: the pad is driven high *before* its output is enabled, so +/// enabling the driver cannot pull the bus low for the few cycles before the peripheral takes over. +/// A low SCL glitch is a clock edge to every device on the bus. +pub fn configurePins(port: u8, scl_pin: u8, sda_pin: u8, opts: struct { + internal_pullups: bool = false, +}) void { + std.debug.assert(port < port_count); + for ([_]struct { pin: u8, signal: u32 }{ + .{ .pin = scl_pin, .signal = sclSignal(port) }, + .{ .pin = sda_pin, .signal = sdaSignal(port) }, + }) |wire| { + gpio.setHigh(wire.pin); + gpio.setInputEnable(wire.pin, true); + gpio.setOpenDrain(wire.pin, true); + gpio.setPull(wire.pin, if (opts.internal_pullups) .up else .none); + gpio.matrixOut(wire.pin, wire.signal); + gpio.matrixIn(wire.pin, wire.signal); + } +} + +// ------------------------------------------------------------------------------- transfers + +/// Bring a port up as a master on a given bus frequency, in the order the hardware requires: +/// clocks, then reset, then configuration, then commit. +/// +/// Reset before configure, because a reset drops everything configured before it. `clkrst.init` +/// does the gate-then-reset pair; the controller clock is separate and enabled after, since it only +/// feeds the state machine. +pub fn init(port: u8, opts: struct { + source: Source = .xtal, + source_hz: u32 = xtal_hz, + bus_hz: u32 = 100_000, + /// Glitch filter width in source-clock cycles. ESP-IDF's driver default is 7. + filter_cycles: u4 = 7, +}) void { + std.debug.assert(port < port_count); + switch (port) { + 0 => clkrst.init(.i2c0), + else => clkrst.init(.i2c1), + } + setControllerClockEnabled(port, true); + setSource(port, opts.source); + + initMaster(port); + setFifoMode(port, true); + disableInterrupts(port); + clearInterrupts(port, all_interrupts); + setBusTiming(port, opts.source_hz, opts.bus_hz); + setFilter(port, opts.filter_cycles); + commitConfig(port); +} + +/// Default spin budget for `write`/`read`. See `waitTransaction`. +pub const default_spins: u32 = 200_000; + +/// Write `bytes` to a 7-bit address as one command list. +/// +/// RSTART | WRITE (1 + len bytes, ack checked) | STOP +/// +/// The address byte goes in the TX FIFO ahead of the data and is counted in the WRITE command's byte +/// count: to the sequencer the address is just the first byte written after a START. `ack_check` is +/// on, so a missing device comes back as `.nack` rather than as a successful write into nothing. +/// +/// One command list, one FIFO load: at most `fifo_len - 1` = 31 data bytes. Longer transfers need +/// the END-and-continue loop that ESP-IDF's driver runs from its interrupt handler, which is out of +/// scope here - hence the assert rather than a partial write. +pub fn write(port: u8, address: u7, bytes: []const u8, spins: u32) Outcome { + std.debug.assert(bytes.len < fifo_len); + resetTxFifo(port); + resetRxFifo(port); + clearInterrupts(port, all_interrupts); + + writeTxFifo(port, &[_]u8{@as(u8, address) << 1}); + writeTxFifo(port, bytes); + + writeCommands(port, &.{ + .{ .op = .restart }, + .{ .op = .write, .bytes = @intCast(bytes.len + 1), .ack_check = true }, + .{ .op = .stop }, + }); + commitConfig(port); + startTransaction(port); + return waitTransaction(port, spins); +} + +/// Read into `out` from a 7-bit address as one command list. +/// +/// RSTART | WRITE 1 (address|read, ack checked) | READ n-1 sending ACK | READ 1 sending NACK | STOP +/// +/// The last byte is a separate command because its ACK bit differs: a master that ACKs the final +/// byte tells the slave to keep going, and the slave then holds SDA for a byte that will never be +/// clocked out. That is the classic I2C read bug, and it is a *command list* bug - which is why the +/// split is here rather than being something the caller can get wrong. +/// +/// Reads of one byte collapse to a single NACKed READ, so the list is four commands instead of five. +pub fn read(port: u8, address: u7, out: []u8, spins: u32) Outcome { + std.debug.assert(out.len > 0); + std.debug.assert(out.len <= fifo_len); + resetTxFifo(port); + resetRxFifo(port); + clearInterrupts(port, all_interrupts); + + writeTxFifo(port, &[_]u8{(@as(u8, address) << 1) | 1}); + + writeCommand(port, 0, .{ .op = .restart }); + writeCommand(port, 1, .{ .op = .write, .bytes = 1, .ack_check = true }); + var slot: u8 = 2; + if (out.len > 1) { + writeCommand(port, slot, .{ .op = .read, .bytes = @intCast(out.len - 1), .ack_value = 0 }); + slot += 1; + } + writeCommand(port, slot, .{ .op = .read, .bytes = 1, .ack_value = 1 }); + writeCommand(port, slot + 1, .{ .op = .stop }); + + commitConfig(port); + startTransaction(port); + const result = waitTransaction(port, spins); + if (result == .complete) readRxFifo(port, out); + return result; +} + +test "the timing arithmetic reproduces ESP-IDF's, including where it looks wrong" { + // 100 kHz on a 40 MHz XTAL: the case every I2C device supports, worked through by hand from + // i2c_ll.h:104-128. clkm_div = 40e6/(100e3*1024) + 1 = 0 + 1 = 1, so sclk stays 40 MHz and + // half = 40e6/100e3/2 = 200. + const t100 = Timing.calculate(40_000_000, 100_000); + try std.testing.expectEqual(@as(u16, 1), t100.clkm_div); + try std.testing.expectEqual(@as(u16, 200), t100.scl_low); + try std.testing.expectEqual(@as(u16, 98), t100.scl_wait_high); // half/2 - 2 + try std.testing.expectEqual(@as(u16, 102), t100.scl_high); // half - wait_high + try std.testing.expectEqual(@as(u16, 50), t100.sda_hold); + try std.testing.expectEqual(@as(u16, 100), t100.sda_sample); + try std.testing.expectEqual(@as(u16, 200), t100.setup); + try std.testing.expectEqual(@as(u16, 200), t100.hold); + // 5*200 = 1000, which needs 10 bits, so 32 - 22 + 2 = 12: a timeout of 2^12 = 4096 cycles, + // 102 us at 40 MHz, about ten bit times. + try std.testing.expectEqual(@as(u16, 12), t100.tout); + + // 400 kHz: same divider, quarter the half-cycle. + const t400 = Timing.calculate(40_000_000, 400_000); + try std.testing.expectEqual(@as(u16, 1), t400.clkm_div); + try std.testing.expectEqual(@as(u16, 50), t400.scl_low); + try std.testing.expectEqual(@as(u16, 23), t400.scl_wait_high); + try std.testing.expectEqual(@as(u16, 27), t400.scl_high); + try std.testing.expectEqual(@as(u16, 10), t400.tout); + + // 10 kHz: the branch that actually uses the controller-clock divider. 40e6/(10e3*1024) = 3, so + // clkm_div = 4, sclk = 10 MHz and half = 500 - just inside the nine-bit period fields, which is + // what the divider exists to guarantee. + const t10 = Timing.calculate(40_000_000, 10_000); + try std.testing.expectEqual(@as(u16, 4), t10.clkm_div); + try std.testing.expectEqual(@as(u16, 500), t10.scl_low); + // Below 80 kHz the wait-high split changes: half/4 rather than half/2 - 2. + try std.testing.expectEqual(@as(u16, 125), t10.scl_wait_high); + try std.testing.expectEqual(@as(u16, 375), t10.scl_high); + + // The hardware ordering constraint IDF asserts (i2c_ll.h:126-127) across the whole range. + for ([_]u32{ 10_000, 50_000, 100_000, 400_000, 1_000_000 }) |hz| { + const t = Timing.calculate(40_000_000, hz); + try std.testing.expect(t.scl_wait_high < t.sda_sample); + try std.testing.expect(t.sda_sample < t.scl_high); + // Every period register is nine bits wide, and scl_low is written minus one. + try std.testing.expect(t.scl_low - 1 <= 511); + try std.testing.expect(t.scl_wait_high <= 127); // this one is seven + try std.testing.expect(t.tout <= 31); // and the timeout exponent is five + } +} + +test "the timeout exponent rounds up, and where the five-bit field runs out" { + // 2000 us at 40 MHz is 80,000 cycles; 2^17 = 131,072 is the first power of two above it, so + // IDF's documented default SCL timeout comes out as 17 - which fits the five-bit field with + // room to spare. This test exists because the first version of this file asserted the opposite. + try std.testing.expectEqual(@as(u32, 17), timeoutExponent(40_000_000, 2000)); + try std.testing.expect(timeoutExponent(40_000_000, 2000) <= time_out_value.max()); + // The field runs out at 2^31 source cycles, 53.7 seconds at 40 MHz - a timeout no I2C bus has a + // use for, which is why neither IDF nor this file range-checks it. Past that the exponent is + // truncated by the field write rather than rejected, exactly as IDF's bitfield store does. + try std.testing.expectEqual(@as(u32, 32), timeoutExponent(40_000_000, 100_000_000)); + try std.testing.expect(timeoutExponent(40_000_000, 100_000_000) > time_out_value.max()); +} + +test "commands encode to the layout i2c_ll_hw_cmd_t describes" { + // A WRITE of three bytes with ACK checking: byte_num=3, ack_en=1, op_code=1. + try std.testing.expectEqual( + @as(u32, 3) | (1 << 8) | (1 << 11), + (Command{ .op = .write, .bytes = 3, .ack_check = true }).encode(), + ); + // RESTART is opcode 6 on this chip, not 0 - the number the register header's prose still gives. + try std.testing.expectEqual(@as(u32, 6 << 11), (Command{ .op = .restart }).encode()); + // A final READ NACKs: ack_val=1 at bit 10, opcode 3. + try std.testing.expectEqual( + @as(u32, 1) | (1 << 10) | (3 << 11), + (Command{ .op = .read, .bytes = 1, .ack_value = 1 }).encode(), + ); + // STOP is 2 and READ is 3, which is the pair the ESP32-era numbering had the other way around. + try std.testing.expectEqual(@as(u32, 2 << 11), (Command{ .op = .stop }).encode()); +} diff --git a/src/hal/intr.zig b/src/hal/intr.zig new file mode 100644 index 0000000..38f8789 --- /dev/null +++ b/src/hal/intr.zig @@ -0,0 +1,965 @@ +//! The interrupt controller. The ESP32-P4 has a **CLIC**, not a PLIC and not the Xtensa-style +//! fixed matrix of the older parts: `soc_caps.h:191` defines SOC_INT_CLIC_SUPPORTED 1, and +//! `soc/interrupt_reg.h:16` says so in prose. Three consequences shape this file. +//! +//! **1. Two independent stages.** A peripheral source does not have a CPU interrupt number; it has +//! a *mapping register*. The interrupt matrix at DR_REG_INTERRUPT_CORE0_BASE holds one 6-bit word +//! per source, and writing `line + 16` into it points that source at external CLIC line `line`. +//! The `+ 16` is not decoration: the CLIC's first 16 IDs are the RISC-V internal interrupts +//! (software, timer, external), so the 32 lines a driver may use are IDs 16..47. +//! `hal/interrupt_clic_ll.h:35-48` is the matrix write; the `+ RV_EXTERNAL_INT_OFFSET` that turns a +//! line number into a CLIC ID is one level up, at `riscv/interrupt_clic.c:26`. Per-line control - +//! enable, trigger, priority, pending - is the *other* stage, in the CLIC's own register file at +//! DR_REG_CLIC_CTRL_BASE, and it is indexed by CLIC ID, i.e. by `line + 16` again. +//! +//! **2. The threshold is a memory-mapped register on this die, not the `mintthresh` CSR.** This is +//! the single easiest thing to get wrong here, because every RISC-V CLIC document and every +//! ESP32-P4 rev-3 build says `mintthresh` (CSR 0x347). `soc/interrupt_reg.h:28-40` selects +//! `INTTHRESH_STANDARD 0` under CONFIG_ESP32P4_SELECTS_REV_LESS_V3 - the same condition that +//! selects the `register/hw_ver1` headers this project builds against - and +//! `riscv/csr_clic.h:37-47` then leaves MINTTHRESH_CSR *undefined*. The threshold lives in +//! CLIC_INT_THRESH_REG at 0x2080_0008, bits [31:24] (`soc/clic_reg.h:61-67`). Writing CSR 0x347 on +//! this silicon is not an illegal instruction and not an error; it writes a register the interrupt +//! arbiter does not read, so interrupts stay masked and nothing says why. +//! +//! **3. `regs.INTTHRESH_STANDARD` lies, and must not be used.** The register module is +//! `zig translate-c` over the headers with *no* sdkconfig, so CONFIG_ESP32P4_SELECTS_REV_LESS_V3 is +//! absent there and `interrupt_reg.h` takes its `#else` branch: the translated module contains +//! `pub const INTTHRESH_STANDARD = 1`, which is the wrong answer for this die. (The oracle's C side +//! is compiled against `src/oracle/oracle_sdkconfig.h:25`, which does define it, so IDF's own code +//! there takes the correct branch. The two disagree, deliberately, and only the C side is right +//! about this macro.) Nothing in this file reads it. +//! +//! Nothing below has been run on hardware by the author of this file. What is claimed is that the +//! register arithmetic matches ESP-IDF's at the cited lines, and that `src/oracle/intr_cases.zig` +//! compares the two on the die. Taking an actual interrupt is a behavioural property no register +//! comparison can establish; see the note at the foot of that file. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +// ------------------------------------------------------------------------------- geometry + +/// CLIC IDs 0..15 are the RISC-V internal interrupts; a driver cannot have them. IDs 16..47 are the +/// 32 external lines. `riscv/csr_clic.h:28-29` (RV_EXTERNAL_INT_COUNT, RV_EXTERNAL_INT_OFFSET) and +/// `soc/clic_reg.h:14` (CLIC_EXT_INTR_NUM_OFFSET) are three names for these two numbers. +pub const line_count: u32 = 32; +pub const ext_offset: u32 = @intCast(regs.CLIC_EXT_INTR_NUM_OFFSET); +/// 16 internal + 32 external. `hal/interrupt_clic_ll.h:22` RV_TOTAL_INT_COUNT, and the hardware +/// agrees: CLIC_INT_INFO_REG's NUM_INT field reads 48 at reset (`soc/clic_reg.h:54-59`). +pub const total_ids: u32 = 48; + +/// Priority levels. `soc/clic_reg.h:13` NLBITS 3, so 8 levels, held in the *top* 3 bits of the +/// 8-bit CLIC_INT_CTL field. Level 0 is masked by the reset threshold; a usable interrupt wants 1 +/// or more. +pub const NLBITS: u5 = @intCast(regs.NLBITS); +const nlbits_shift: u5 = 8 - NLBITS; +/// The low `8 - NLBITS` bits of a priority/threshold byte are not part of the level and IDF fills +/// them with ones (`riscv/csr_clic.h:59`, NLBITS_TO_BYTE). Reproduced exactly, because the +/// differential compares the whole word. +const nlbits_pad: u32 = (@as(u32, 1) << nlbits_shift) - 1; + +// -------------------------------------------------------------------------- interrupt matrix + +/// Every peripheral interrupt source on this chip, from `soc/interrupts.h` - which opens with +/// "This table is decided by hardware, don't touch this." +/// +/// IDs 0..127 are contiguous and each has a mapping register at `matrix_base + 4*id`: the last of +/// them, `assist_debug` = 127, is INTERRUPT_CORE0_ASSIST_DEBUG_INT_MAP_REG at +0x1FC, which is +/// exactly 4*127. That is the invariant `interrupt_clic_ll.h:46` depends on when it computes the +/// address arithmetically rather than from a table. +/// +/// **The last three exist only on chip revision >= 3.0 and therefore not on this die.** +/// `soc/interrupts.h:155-160` explains the gap: their mapping registers are *not* contiguous with +/// the rest, so IDF gave them IDs 133-135 to make `base + 4*id` land on the right address anyway. +/// The numbering hole at 128..132 is that workaround, not missing hardware. On a pre-v3 part - +/// which is what `regs.ZIG_P4_HW_VER == 1` asserts - routing one of them writes a register that +/// nothing drives. +pub const Source = enum(u8) { + lp_rtc = 0, + lp_wdt = 1, + lp_timer_reg0 = 2, + lp_timer_reg1 = 3, + mb_hp = 4, + mb_lp = 5, + pmu_0 = 6, + pmu_1 = 7, + lp_anaperi = 8, + lp_adc = 9, + lp_gpio = 10, + lp_i2c = 11, + lp_i2s = 12, + lp_spi = 13, + lp_touch = 14, + /// Also spelled ETS_TEMPERATURE_SENSOR_INTR_SOURCE; IDF aliases the two (`interrupts.h:34`). + lp_tsens = 15, + lp_uart = 16, + lp_efuse = 17, + lp_sw = 18, + lp_sysreg = 19, + lp_huk = 20, + sys_icm = 21, + usb_serial_jtag = 22, + sdio_host = 23, + dw_gdma = 24, + spi2 = 25, + spi3 = 26, + i2s0 = 27, + i2s1 = 28, + i2s2 = 29, + uhci0 = 30, + uart0 = 31, + uart1 = 32, + uart2 = 33, + uart3 = 34, + uart4 = 35, + lcd_cam = 36, + adc = 37, + pwm0 = 38, + pwm1 = 39, + twai0 = 40, + twai1 = 41, + twai2 = 42, + rmt = 43, + i2c0 = 44, + i2c1 = 45, + tg0_t0 = 46, + tg0_t1 = 47, + tg0_wdt_level = 48, + tg1_t0 = 49, + tg1_t1 = 50, + tg1_wdt_level = 51, + ledc = 52, + systimer_target0 = 53, + systimer_target1 = 54, + systimer_target2 = 55, + ahb_pdma_in_ch0 = 56, + ahb_pdma_in_ch1 = 57, + ahb_pdma_in_ch2 = 58, + ahb_pdma_out_ch0 = 59, + ahb_pdma_out_ch1 = 60, + ahb_pdma_out_ch2 = 61, + axi_pdma_in_ch0 = 62, + axi_pdma_in_ch1 = 63, + axi_pdma_in_ch2 = 64, + axi_pdma_out_ch0 = 65, + axi_pdma_out_ch1 = 66, + axi_pdma_out_ch2 = 67, + rsa = 68, + aes = 69, + sha = 70, + ecc = 71, + ecdsa = 72, + km = 73, + gpio_intr0 = 74, + gpio_intr1 = 75, + gpio_intr2 = 76, + gpio_intr3 = 77, + gpio_pad_comp = 78, + from_cpu_intr0 = 79, + from_cpu_intr1 = 80, + from_cpu_intr2 = 81, + from_cpu_intr3 = 82, + cache = 83, + mspi = 84, + csi_bridge = 85, + dsi_bridge = 86, + csi = 87, + dsi = 88, + gmii_phy = 89, + lpi = 90, + pmt = 91, + eth_mac = 92, + usb_otg = 93, + usb_otg_endp_multi_proc = 94, + jpeg = 95, + ppa = 96, + core0_trace = 97, + core1_trace = 98, + hp_core_ctrl = 99, + isp = 100, + i3c_mst = 101, + i3c_slv = 102, + usb_otg11_ch0 = 103, + dma2d_in_ch0 = 104, + dma2d_in_ch1 = 105, + dma2d_out_ch0 = 106, + dma2d_out_ch1 = 107, + dma2d_out_ch2 = 108, + psram_mspi = 109, + hp_sysreg = 110, + pcnt = 111, + hp_pau = 112, + hp_parlio_rx = 113, + hp_parlio_tx = 114, + h264_dma2d_out_ch0 = 115, + h264_dma2d_out_ch1 = 116, + h264_dma2d_out_ch2 = 117, + h264_dma2d_out_ch3 = 118, + h264_dma2d_out_ch4 = 119, + h264_dma2d_in_ch0 = 120, + h264_dma2d_in_ch1 = 121, + h264_dma2d_in_ch2 = 122, + h264_dma2d_in_ch3 = 123, + h264_dma2d_in_ch4 = 124, + h264_dma2d_in_ch5 = 125, + h264_reg = 126, + assist_debug = 127, + + /// Chip rev >= 3.0 only - absent on this die. See the note above. + dma2d_in_ch2 = 133, + /// Chip rev >= 3.0 only - absent on this die. + dma2d_out_ch3 = 134, + /// Chip rev >= 3.0 only - absent on this die. + axi_perf_mon = 135, + + /// True on a source that this pre-v3 silicon does not have. + pub inline fn isRev3Only(self: Source) bool { + return @intFromEnum(self) >= 133; + } +}; + +/// The last source ID with a mapping register on pre-v3 silicon. +pub const max_source_id: u8 = @intFromEnum(Source.assist_debug); + +/// Core 0's interrupt matrix. Core 1's is 0x800 above it (`reg_base.h:198-199`) and is not reachable +/// from here: this image runs core 0 only - core 1 is held in reset at power-on +/// (HP_SYS_CLKRST REG_RST_EN_CORE1_GLOBAL defaults to 1) - and routing a source to a core that is +/// not running is a way to lose an interrupt silently rather than loudly. +const matrix_base: u32 = mmio.addr(regs.DR_REG_INTERRUPT_CORE0_BASE); + +/// The mapping register's only field: 6 bits, holding a CLIC ID. Taken from UART0's macro pair +/// because the field is identical in all 128 of them - `interrupt_core0_reg.h` repeats +/// `_INT_MAP` / mask 0x3F / shift 0 for every source. (`INTERRUPT_CORE0_*_INT_MAP_M` is one of the +/// 153 `_M` macros that are broken C inside ESP-IDF and appear here as poisoned decls; the `_S`/`_V` +/// pair is the only usable form, which is what `mmio.Field.of` takes.) +const int_map = Field.of(regs.INTERRUPT_CORE0_UART0_INT_MAP_S, regs.INTERRUPT_CORE0_UART0_INT_MAP_V); + +inline fn mapReg(source_id: u8) Reg { + return Reg.atAddress(matrix_base + 4 * @as(u32, source_id)); +} + +/// Point a peripheral source at an external CLIC line. +/// +/// This is only the matrix half. A routed source still needs `setEnabled(line, true)`, a trigger +/// type, a priority above the threshold, a handler, and mstatus.MIE - `configureLine` does the +/// CLIC-side four in the order the hardware wants. +/// +/// Several sources may share one line; that is the normal way to fit 128 sources into 32 lines, and +/// the handler then has to ask each peripheral whether it was the one. Nothing here prevents it. +pub fn route(source: Source, line: u5) void { + routeId(@intFromEnum(source), line); +} + +/// `route` by raw source ID, for a source this enum does not name. +/// +/// The write is a read-modify-write of the low 6 bits, exactly as `interrupt_clic_ll.h:46` does it +/// (`REG_SET_BITS(DR_REG_INTERRUPT_CORE0_BASE + 4*intr_src, intr_num, RV_INT_MASK)` with +/// RV_INT_MASK 63 at line 25). The upper 26 bits are reserved and preserved. +pub fn routeId(source_id: u8, line: u5) void { + std.debug.assert(source_id <= max_source_id); + mapReg(source_id).modify(.{int_map.is(@as(u32, line) + ext_offset)}); +} + +/// Detach a source from every line. +/// +/// Writes CLIC ID 0, which is `ETS_INVALID_INUM` on this chip (`soc/esp32p4/include/soc/soc.h:251`) +/// and is what `esp_system/port/cpu_start.c:185` writes into all 128 mapping registers at boot. +/// ID 0 is an internal RISC-V interrupt line that the matrix cannot actually drive, so it means +/// "nowhere" rather than "line 0" - note the asymmetry with `route`, which adds 16. +pub fn unroute(source: Source) void { + mapReg(@intFromEnum(source)).modify(.{int_map.is(0)}); +} + +/// Which external line a source is routed to, or null if it is unrouted or points at an internal ID. +pub fn routedLine(source: Source) ?u5 { + const id = mapReg(@intFromEnum(source)).get(int_map); + if (id < ext_offset or id >= ext_offset + line_count) return null; + return @intCast(id - ext_offset); +} + +// ------------------------------------------------------------------------- per-line control + +/// One 32-bit control word per CLIC ID at `DR_REG_CLIC_CTRL_BASE + 4*id` (`soc/clic_reg.h:69`). +/// Indexed by CLIC ID, so every accessor here adds `ext_offset` to the caller's line number. +/// +/// The same word is also described byte-wise by the `BYTE_CLIC_*` macros (clic_reg.h:113-160), and +/// ESP-IDF uses both spellings: `interrupt_clic_ll.h` does 32-bit REG_SET_FIELD, the TEE build does +/// 8-bit stores. They land on the same bits, and each field sits wholly inside one byte, so a +/// 32-bit read-modify-write of one field and a byte store of that byte are indistinguishable in the +/// resulting word. This file uses the 32-bit form throughout. +const clic_ctrl_base: u32 = mmio.addr(regs.DR_REG_CLIC_CTRL_BASE); + +/// Priority, bits [31:24]. Reset value 0x1f (clic_reg.h:70). +const int_ctl = Field.of(regs.CLIC_INT_CTL_S, regs.CLIC_INT_CTL_V); +/// Trigger type, bits [18:17]. +const int_attr_trig = Field.of(regs.CLIC_INT_ATTR_TRIG_S, regs.CLIC_INT_ATTR_TRIG_V); +/// Hardware vectoring: 1 means fetch the handler address from MTVT rather than trapping to mtvec. +const int_attr_shv = Field.of(regs.CLIC_INT_ATTR_SHV_S, regs.CLIC_INT_ATTR_SHV_V); +/// Enable, bit 8. +const int_ie = Field.of(regs.CLIC_INT_IE_S, regs.CLIC_INT_IE_V); +/// Pending, bit 0. Read/write, with asymmetric semantics - see `edgeAck`. +const int_ip = Field.of(regs.CLIC_INT_IP_S, regs.CLIC_INT_IP_V); + +inline fn ctrl(line: u5) Reg { + return Reg.atAddress(clic_ctrl_base + 4 * (@as(u32, line) + ext_offset)); +} + +/// By raw CLIC ID rather than by external line, for the one caller that has to reach the 16 +/// internal IDs: `init`, silencing everything the ROM may have left enabled. +inline fn ctrlRegById(clic_id: u32) Reg { + std.debug.assert(clic_id < total_ids); + return Reg.atAddress(clic_ctrl_base + 4 * clic_id); +} + +/// How a source drives its line. The encoding is a two-bit field whose *low* bit selects +/// level-versus-edge and whose high bit selects the edge, which is why `interrupt_clic_ll.h:60` +/// masks the read with `& 1` to answer "is it edge-triggered": `0b10` is a level interrupt too. +/// (`soc/clic_reg.h:84-88`.) +pub const Trigger = enum(u2) { + level = 0, + rising_edge = 1, + /// 0b10 - low bit clear, so this is a *level* trigger despite the encoding's shape. Present + /// only because the field is two bits wide; no source should be configured with it. + level_alias = 2, + falling_edge = 3, + + pub inline fn isEdge(self: Trigger) bool { + return @intFromEnum(self) & 1 != 0; + } +}; + +pub fn setEnabled(line: u5, on: bool) void { + ctrl(line).modify(.{int_ie.is(@intFromBool(on))}); +} + +pub fn isEnabled(line: u5) bool { + return ctrl(line).get(int_ie) == 1; +} + +pub fn setTrigger(line: u5, t: Trigger) void { + ctrl(line).modify(.{int_attr_trig.is(@intFromEnum(t))}); +} + +pub fn getTrigger(line: u5) Trigger { + return @enumFromInt(ctrl(line).get(int_attr_trig)); +} + +/// Priority 0..7, stored left-aligned in the 8-bit CLIC_INT_CTL field. +/// +/// The stored byte is `priority << (8 - NLBITS)` with the low bits **zero**, which is what +/// `esp_tee_rv_utils.h:112` writes and what `interrupt_clic_ll.h:74` reads back with `>> (8-NLBITS)`. +/// Note the asymmetry with the *threshold*, where IDF fills the same low bits with ones +/// (`csr_clic.h:59`). Copying the threshold's encoding here would leave a different word behind +/// than IDF's, for the same nominal priority. +pub fn setPriority(line: u5, priority: u3) void { + ctrl(line).modify(.{int_ctl.is(@as(u32, priority) << nlbits_shift)}); +} + +pub fn getPriority(line: u5) u3 { + return @intCast(ctrl(line).get(int_ctl) >> nlbits_shift); +} + +/// Hardware vectoring for one line. With SHV set, the CLIC jumps to `MTVT + 4*id` instead of to +/// mtvec's base; `installVectorTable` fills every slot with the same trap entry, so flipping this +/// changes the fetch path and not the code that runs. `interrupt_clic_ll.h:99-102`. +pub fn setVectored(line: u5, on: bool) void { + ctrl(line).modify(.{int_attr_shv.is(@intFromBool(on))}); +} + +pub fn isVectored(line: u5) bool { + return ctrl(line).get(int_attr_shv) == 1; +} + +pub fn isPending(line: u5) bool { + return ctrl(line).get(int_ip) == 1; +} + +/// Acknowledge an edge-triggered interrupt. +/// +/// Writing **1** to IP is what clears it for an edge source. That reads backwards, and clic_reg.h +/// only hints at it - "This bit has different set and clear logic in the case of level interrupt +/// and edge interrupt" (clic_reg.h:106-107) - but ESP-IDF's function that does exactly this store is +/// named `rv_utils_intr_edge_ack` (`esp_private/interrupt_clic.h`, the `REG_SET_BIT(..., CLIC_INT_IP)` +/// at the end of that header). For a *level* source this instead asserts the pending bit, which is +/// how software raises one by hand; there is no acknowledge for a level source at the CLIC at all, +/// the handler must clear the peripheral's own status register. +pub fn edgeAck(line: u5) void { + ctrl(line).modify(.{int_ip.is(1)}); +} + +/// Raise a line from software. Same store as `edgeAck`; the two names exist because the hardware +/// gives one write two meanings depending on `Trigger`. +pub fn setPending(line: u5) void { + ctrl(line).modify(.{int_ip.is(1)}); +} + +/// Bitmask of the 32 external lines that are enabled, one loop over the control words. Mirrors +/// `rv_utils_intr_get_enabled_mask` in `esp_private/interrupt_clic.h`. +pub fn enabledMask() u32 { + var m: u32 = 0; + var i: u5 = 0; + while (true) : (i += 1) { + if (isEnabled(i)) m |= @as(u32, 1) << i; + if (i == line_count - 1) break; + } + return m; +} + +// ----------------------------------------------------------------------------- the threshold + +/// CLIC_INT_THRESH_REG - 0x2080_0008 (`soc/clic_reg.h:61`), **not** the `mintthresh` CSR. See the +/// module comment: on this pre-v3 die `csr_clic.h` does not even define MINTTHRESH_CSR, and a write +/// to CSR 0x347 here is accepted and ignored. +const thresh_reg = Reg.at(regs.CLIC_INT_THRESH_REG); +const cpu_int_thresh = Field.of(regs.CLIC_CPU_INT_THRESH_S, regs.CLIC_CPU_INT_THRESH_V); + +/// Mask every interrupt whose priority is <= `level`. +/// +/// The comparison is **inclusive**: threshold 0 lets priorities 1..7 through, threshold 7 masks +/// everything. `esp_private/interrupt_clic.h:198-203` makes the same point when it computes +/// `mask_int_level_lower_than(n)` as `set_intlevel(n - 1)`. Reset is 0, i.e. open. +/// +/// Two details reproduced from IDF rather than invented: +/// * the byte is `(level << 5) | 0x1f` - the low `8 - NLBITS` bits are filled with **ones** +/// (`csr_clic.h:59`, NLBITS_TO_BYTE), which is the opposite of the per-line priority encoding; +/// * the register is read back immediately afterwards. That is not a paranoid verification, it is +/// ordering: `esp_private/interrupt_clic.h:139-144` records that the CPU does not see the new +/// threshold until the store has actually left the write buffer, and that a load - or about +/// eight nops - is what forces it. Without the load, re-enabling mstatus.MIE on the next +/// instruction can take an interrupt the new threshold was meant to mask. +/// +/// `write` rather than `modify` is deliberate and matches IDF's `REG_WRITE`: CLIC_CPU_INT_THRESH is +/// the register's only field, so there is nothing to preserve. +pub fn setThreshold(level: u3) void { + thresh_reg.write(.{cpu_int_thresh.is((@as(u32, level) << nlbits_shift) | nlbits_pad)}); + _ = thresh_reg.raw(); +} + +pub fn getThreshold() u3 { + return @intCast(thresh_reg.get(cpu_int_thresh) >> nlbits_shift); +} + +// ------------------------------------------------------------- vector table and trap entry + +/// CSR numbers, from `components/riscv/include/riscv/csr_clic.h`: +/// * `MTVT_CSR 0x307` (line 34) - base of the interrupt jump table. +/// * `MTVEC_MODE_CSR 3` (line 22) - the two low bits of mtvec that put the core in CLIC mode. +/// * `MINTSTATUS_CSR 0x346` (`soc/interrupt_reg.h:36`) - **non-standard on this die**; the RISC-V +/// CLIC specification and IDF's rev-3 path both say 0xFB1 (`csr_clic.h:40`). +/// * `MINTTHRESH_CSR 0x347` exists only when INTTHRESH_STANDARD is 1, which it is not here. +pub const mtvt_csr = 0x307; +pub const mintstatus_csr = 0x346; +pub const mtvec_mode_clic = 3; +/// mstatus.MIE. Same bit `clkrst.Guard` manipulates. +const mstatus_mie: u32 = 1 << 3; + +/// A line's handler. Runs with mstatus.MIE clear - this file does not implement nesting - on the +/// interrupted stack, so it must be short and must not use floating point: `trapEntry` saves the +/// integer caller-saved registers and nothing else, and `_start` leaves the FPU enabled, so a +/// handler that touches an f-register corrupts whatever it interrupted. +pub const Handler = *const fn (line: u5) void; + +var handlers: [line_count]?Handler = @splat(null); + +/// Interrupts that arrived on a line with no handler, or on one of the 16 internal CLIC IDs. Not +/// reset by anything here: a non-zero value after a run is the diagnostic. +pub var spurious: u32 = 0; + +/// The CLIC's jump table: one address per CLIC ID, internal and external. +/// +/// 48 entries, and 256-byte aligned because the CLIC requires MTVT to be aligned to a power of two +/// at least as large as the table (4 * 48 = 192 bytes, so 256). The alignment travels with the +/// symbol, so the generated linker script's `.bss ... ALIGN(4)` is not a problem - the linker pads +/// to the input section's own alignment. No dedicated section is needed and build.zig is unchanged. +/// +/// Every slot points at the same `trapEntry`. A per-line stub would save the dispatch load, but it +/// would be 48 near-identical pieces of assembly to be wrong in, and the win is a handful of cycles +/// against a handler call. The table exists because the hardware needs one when SHV is set, not +/// because the entries differ. +var vector_table: [total_ids]u32 align(256) = @splat(0); + +/// What `init` found before it changed anything. Diagnostics, and the only record of the state the +/// bootloader hands over in - every one of these is overwritten by `init` itself, so nothing else +/// can observe them. +pub var boot_state: BootState = .{}; +pub const BootState = struct { + /// mstatus.MIE as handed over. Measured 1 on this board, which is the fact the whole ownership + /// sequence below exists for. + mie: bool = false, + /// Which of the 32 external lines had CLIC_INT_IE set before `init` cleared them. + enabled_lines: u32 = 0, + /// How many of the 128 peripheral sources were pointing at an external line before `init` + /// detached them. + routed_sources: u32 = 0, +}; + +/// Take ownership of the interrupt controller, then point it at this file. +/// +/// **The bootloader hands over with interrupts globally enabled.** Measured: `mie_at_boot=1`. That +/// single fact is why this function is a sequence rather than three CSR writes, and it cost two +/// silent hangs to establish. Two separate hazards follow from it, and clearing MIE only fixes the +/// first: +/// +/// 1. `init(); attach(...)` used to take an interrupt the moment the line's IE bit went up, before +/// the caller had said it was ready. `globalDisable()` first fixes that. +/// +/// 2. **Whatever the ROM had armed is still armed.** The ROM ran with its own mtvec and its own +/// reasons to enable interrupts; the matrix and the CLIC's IE bits are not reset by the handover. +/// The instant this file's caller sets MIE, any line the ROM left enabled vectors into +/// `trapEntry` - on an ID nothing here has a handler for. That increments `spurious` and +/// `mret`s; and if the source is level-triggered and still asserting, the next instruction traps +/// again, forever, with the console silent. The failure looks exactly like "our own line is not +/// being delivered", which is what it was mistaken for. +/// +/// So this function does what ESP-IDF's `core_intr_matrix_clear` does before it trusts the +/// controller (`esp_system/port/cpu_start.c:174-198`), and in the same order: +/// * detach all 128 sources by writing ETS_INVALID_INUM (cpu_start.c:183-189); +/// * clear every line's enable, which IDF gets for free from the CLIC's reset values and this +/// image does not, because the ROM ran first; +/// * set every external line vectored (cpu_start.c:193-196 - "Set all the CPU interrupt lines to +/// vectored by default, as it is on other RISC-V targets"). +/// +/// The register differential could not have found any of this: MIE is a CSR, and the boot state of +/// the matrix is identical on both sides of every comparison because both sides inherit it. +/// +/// Leaves MIE clear. Enabling interrupts stays the caller's decision, via `globalEnable()`. +pub fn init() void { + boot_state.mie = globalEnabled(); + globalDisable(); + + // Record and then silence every line, before anything can be delivered anywhere. + var l: u5 = 0; + while (true) : (l += 1) { + if (isEnabled(l)) boot_state.enabled_lines |= @as(u32, 1) << l; + if (l == line_count - 1) break; + } + // All 48 IDs, internal ones included: this core's interrupts are ours now, and an internal ID + // left enabled is as capable of trapping into `trapEntry` as an external one. + var id: u32 = 0; + while (id < total_ids) : (id += 1) { + ctrlRegById(id).modify(.{int_ie.is(0)}); + } + + // Detach every source. cpu_start.c:183-189 writes ETS_INVALID_INUM (0) to all of them. + var src: u32 = 0; + while (src <= max_source_id) : (src += 1) { + const r = mapReg(@intCast(src)); + const was = r.get(int_map); + if (was >= ext_offset and was < ext_offset + line_count) boot_state.routed_sources += 1; + r.modify(.{int_map.is(0)}); + } + + const entry = @intFromPtr(&trapEntry); + for (&vector_table) |*slot| slot.* = @intCast(entry); + + asm volatile ("csrw %[csr], %[val]" + : + : [csr] "i" (mtvt_csr), + [val] "r" (@as(u32, @intCast(@intFromPtr(&vector_table)))), + ); + // mtvec = base | 3. Mode 3 is what `rv_utils_set_mtvec` writes (`riscv/rv_utils.h:168-171` with + // MTVEC_MODE_CSR from `csr_clic.h:22`) and it is what makes the core interpret mcause and MTVT + // as CLIC rather than as the standard vectored interface. + // + // The hardware uses `mtvec[31:6] << 6` (vectors_clic.S:38-46 spells this out), so it ignores the + // low six bits entirely: a `trapEntry` that were not 64-byte aligned would silently vector up to + // 60 bytes *before* the function. `trapEntryAddress()` exists so a test can prove on the die + // that it is aligned rather than trusting the linker. + asm volatile ("csrw mtvec, %[val]" + : + : [val] "r" (@as(u32, @intCast(entry)) | mtvec_mode_clic), + ); + + // Every external line vectored, matching cpu_start.c:193-196. Also the safer default in its own + // right: SHV=1 is the only delivery path ESP-IDF exercises on this chip, so it is the only one + // the silicon has been validated against. See `configureLine`. + l = 0; + while (true) : (l += 1) { + setVectored(l, true); + if (l == line_count - 1) break; + } + + // Threshold open, matching IDF's RVHAL_INTR_ENABLE_THRESH of 0 (`csr_clic.h:16`): every line + // then gates on its own IE bit and its priority, which is where a driver can reason about it. + setThreshold(0); +} + +/// Diagnostics a behavioural test can print, because the two facts they establish - that the trap +/// entry is 64-byte aligned and that MTVT is 256-byte aligned - are properties of the *link*, and +/// the shipped image is stripped, so there is no way to check them from the host. +pub fn trapEntryAddress() u32 { + return @intCast(@intFromPtr(&trapEntry)); +} + +pub fn vectorTableAddress() u32 { + return @intCast(@intFromPtr(&vector_table)); +} + +pub fn readMtvec() u32 { + return asm volatile ("csrr %[out], mtvec" + : [out] "=r" (-> u32), + ); +} + +pub fn readMtvt() u32 { + return asm volatile ("csrr %[out], %[csr]" + : [out] "=r" (-> u32), + : [csr] "i" (mtvt_csr), + ); +} + +/// mintstatus, CSR 0x346 on this die (`soc/interrupt_reg.h:36`). Bits [31:24] are the current +/// interrupt level: non-zero outside a handler would mean a previous trap never returned. +pub fn readMintstatus() u32 { + return asm volatile ("csrr %[out], %[csr]" + : [out] "=r" (-> u32), + : [csr] "i" (mintstatus_csr), + ); +} + +/// Install (or, with null, remove) the handler for one external line. +/// +/// Done with interrupts masked because the store is a pointer the trap entry may be about to load; +/// `clkrst.maskInterrupts` composes - it restores only the MIE that was there - so this is safe to +/// call from inside an already-masked region. +pub fn setHandler(line: u5, handler: ?Handler) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + handlers[line] = handler; +} + +/// Everything one line needs, in the order the hardware wants: handler before enable, so a source +/// that is already pending cannot reach an empty slot; trigger and priority before enable, so the +/// first interrupt is taken under the intended configuration rather than under the reset one. +/// +/// Does not touch the matrix - `route` is the other half - and does not touch mstatus. +pub fn configureLine(line: u5, opts: struct { + handler: Handler, + trigger: Trigger = .level, + /// Must exceed the threshold to ever be taken; the threshold comparison is inclusive. + priority: u3 = 1, + /// Hardware vectoring: fetch the handler address from `MTVT + 4*id` instead of trapping to + /// mtvec's base. + /// + /// **On by default, and the default is the interesting part.** Every slot of the table holds the + /// same `trapEntry`, so this changes only how the core finds that address - which makes the + /// choice look free, and it is not. ESP-IDF sets SHV on all 32 lines at boot + /// (`cpu_start.c:193-196`, "Set all the CPU interrupt lines to vectored by default, as it is on + /// other RISC-V targets") and puts nothing but `j _panic_handler` at mtvec's base + /// (`vectors_clic.S:47-52`). So on this chip the SHV=0 delivery path is one ESP-IDF never takes + /// and therefore one nobody has validated. Defaulting to the path the vendor exercises is worth + /// more than the memory fetch it costs. + /// + /// **Measured on the die: it is the other way round, and the default is now `false`.** + /// + /// With SHV=1 the interrupt was never delivered. The core vectored to a wild address and took an + /// instruction access fault - `mcause=0x30000001` (EXCCODE 1, MINHV clear, so the fault was not + /// during the table fetch), at a `mepc` that differed run to run, with `taken=0` proving the + /// trap entry was never reached. mtvec, MTVT and the table contents were all verified correct + /// beforehand: `mtvec=0x40001383` = entry|3, `mtvt=0x4ff00100`, and every slot holding + /// `0x40001380` = `trapEntry`. + /// + /// The difference from ESP-IDF is *where the table lives*. IDF's `_mtvt_table` is in + /// `.section .exception_vectors_table.text` (`vectors_clic.S:32,67`), i.e. instruction space. + /// This image has no IRAM: it executes from flash through the MMU, so a table that `init()` has + /// to write must live in L2MEM, and the hardware vector fetch does not appear to work from + /// there. Since flash is not writable at run time, there is nowhere else to put it, which makes + /// SHV=0 the correct choice for this memory layout rather than a workaround. + /// + /// With SHV=0 both halves of the behavioural test pass: one interrupt taken, dispatched to the + /// right handler, `last_clic_id=21`, no spurious - and the threshold experiment then shows the + /// memory-mapped register at 0x2080_0008 really is the one the arbiter reads. + /// + /// `true` remains available for an image that gains an IRAM section, and the vector table is + /// still populated so that switching is a one-word change. + vectored: bool = false, +}) void { + setHandler(line, opts.handler); + setTrigger(line, opts.trigger); + setPriority(line, opts.priority); + setVectored(line, opts.vectored); + setEnabled(line, true); +} + +/// Route a source and bring its line up in one call. +pub fn attach(source: Source, line: u5, opts: struct { + handler: Handler, + trigger: Trigger = .level, + priority: u3 = 1, + /// See `configureLine`: vectored is the only path ESP-IDF exercises on this chip. + vectored: bool = false, +}) void { + route(source, line); + configureLine(line, .{ + .handler = opts.handler, + .trigger = opts.trigger, + .priority = opts.priority, + .vectored = opts.vectored, + }); +} + +// --------------------------------------------------------------------------- global enable + +/// mstatus.MIE on. Nothing is taken before this, whatever the CLIC is configured to do. +pub inline fn globalEnable() void { + asm volatile ("csrs mstatus, %[m]" + : + : [m] "r" (mstatus_mie), + ); +} + +pub inline fn globalDisable() void { + asm volatile ("csrc mstatus, %[m]" + : + : [m] "r" (mstatus_mie), + ); +} + +pub inline fn globalEnabled() bool { + const s = asm volatile ("csrr %[out], mstatus" + : [out] "=r" (-> u32), + ); + return s & mstatus_mie != 0; +} + +/// The composable form: mask, do something, restore whatever was there. +/// +/// const guard = intr.mask(); +/// defer guard.release(); +/// +/// This is `clkrst.maskInterrupts` under another name, re-exported rather than reimplemented so +/// that a critical section written against either module is the same critical section. It nests +/// correctly - `release` only sets MIE if MIE was set on entry - which is why `setHandler` can use +/// it without caring who called it. +pub const Guard = clkrst.Guard; +pub inline fn mask() Guard { + return clkrst.maskInterrupts(); +} + +// ------------------------------------------------------------------------------- trap entry + +/// How many times `trapEntry` has dispatched an interrupt, and the last CLIC ID it saw. Diagnostics: +/// with `taken == 0` the trap was never reached at all, which separates "the CLIC did not deliver" +/// from "the handler did not run". +pub var taken: u32 = 0; +pub var last_clic_id: u32 = 0; + +/// An exception - not an interrupt - that reached `trapEntry`. +pub const Fault = struct { + /// Full mcause. Bit 31 is clear by construction here; the low bits are the exception code + /// (1 instruction access, 2 illegal instruction, 5 load access, 7 store access, 11 ecall). + mcause: u32, + /// The instruction that faulted. + mepc: u32, + /// The address or instruction word involved, per exception code. + mtval: u32, +}; + +pub var faults: u32 = 0; +pub var last_fault: Fault = .{ .mcause = 0, .mepc = 0, .mtval = 0 }; + +/// Called with the fault already recorded, before parking. Install one to get the numbers out; +/// `hal` cannot print, so this hook is the only way a fault becomes visible. +/// +/// hal.intr.on_fault = struct { +/// fn f(x: hal.intr.Fault) void { +/// soc.rom.print("MARK FAULT mcause=0x%08x mepc=0x%08x mtval=0x%08x\r\n", +/// .{ x.mcause, x.mepc, x.mtval }); +/// } +/// }.f; +pub var on_fault: ?*const fn (Fault) void = null; + +/// Called from `trapEntry` with the CLIC ID out of mcause. Not part of the API; `export` because +/// the assembly calls it by name. +export fn intrDispatch(clic_id: u32) callconv(.c) void { + taken +%= 1; + last_clic_id = clic_id; + if (clic_id < ext_offset or clic_id >= ext_offset + line_count) { + // One of the 16 internal IDs. This file routes nothing there, so it is a bug elsewhere - + // most likely something the ROM left armed that `init` did not manage to silence. + spurious +%= 1; + return; + } + const line: u5 = @intCast(clic_id - ext_offset); + if (handlers[line]) |h| h(line) else spurious +%= 1; +} + +/// The exception arm of `trapEntry`. Records, reports if a hook is installed, and **parks**. +/// +/// Parking rather than returning is the whole point. `mret` from an exception resumes at the +/// faulting instruction, which faults again immediately: every mistake anywhere in this file used to +/// become an unbreakable loop through the trap entry with the console silent, indistinguishable from +/// "the interrupt was never delivered". It cost a debugging round to tell those apart. ESP-IDF makes +/// the same choice by putting `j _panic_handler` at mtvec's base (`vectors_clic.S:47-52`). +export fn intrFault(mcause: u32, mepc: u32, mtval: u32) callconv(.c) noreturn { + faults +%= 1; + last_fault = .{ .mcause = mcause, .mepc = mepc, .mtval = mtval }; + globalDisable(); + if (on_fault) |f| f(last_fault); + while (true) {} +} + +/// The trap entry: every trap on this core arrives here, interrupt or exception. +/// +/// Reached three ways, and they are not interchangeable: +/// * an **interrupt with SHV = 1**, through `MTVT + 4*id`; +/// * an **interrupt with SHV = 0**, through mtvec's base; +/// * an **exception**, always through mtvec's base, whatever any line's SHV says. +/// +/// 64-byte aligned, and this is a hardware requirement rather than tidiness: in CLIC mode the core +/// computes the target as `mtvec[31:6] << 6` (`vectors_clic.S:38-46` states it outright), so the low +/// six bits of mtvec are not part of the address. A trap entry that were not 64-byte aligned would +/// vector up to 60 bytes *before* this function, into whatever the linker put there. Measured in the +/// linked image: 0x4000_1140, and `trapEntryAddress()` lets a test confirm it on the die, since the +/// shipped image is stripped and there is no symbol to check from the host. +/// +/// **The first thing it does is decide whether this was an interrupt at all.** mcause bit 31 says +/// so, and getting that wrong is not a small bug: an exception whose handler `mret`s resumes at the +/// faulting instruction and faults again, immediately and forever, with the console silent. That +/// failure is indistinguishable from "the interrupt was never delivered", and the two were in fact +/// confused for a debugging round. So the exception arm never returns - see `intrFault`. +/// +/// Saves the integer caller-saved set - ra, t0-t6, a0-a7, sixteen words - and nothing else. Not +/// saved, deliberately and with consequences: +/// * **the f registers.** `src/main.zig`'s `_start` sets mstatus.FS to enable the FPU, so a handler +/// that does float arithmetic silently corrupts the interrupted code. Handlers must stay integer. +/// * **mepc, mcause, mstatus.** In CLIC mode the core stacks the previous privilege, interrupt +/// enable and interrupt level in mcause itself, and `mret` restores them from there - so nothing +/// here may write mcause, and nothing does. They are only at risk from a *nested* trap, and MIE +/// stays clear for the whole sequence, so nothing can nest. That is also why there is no `mnxti` +/// loop: the CLIC's hardware nesting (SOC_INT_HW_NESTED_SUPPORTED, `soc_caps.h:193`) is unused. +/// +/// One consequence of `mret` worth stating because it defeats an obvious defence: it restores +/// mstatus.MIE from MPIE, which the hardware set to 1 on entry. A handler that calls +/// `globalDisable()` therefore does **not** leave interrupts off after it returns. To stop a runaway +/// source the handler must clear it at the peripheral, or call `setEnabled(line, false)`. +export fn trapEntry() align(64) callconv(.naked) noreturn { + asm volatile ( + \\ addi sp, sp, -64 + \\ sw ra, 0(sp) + \\ sw t0, 4(sp) + \\ sw t1, 8(sp) + \\ sw t2, 12(sp) + \\ sw a0, 16(sp) + \\ sw a1, 20(sp) + \\ sw a2, 24(sp) + \\ sw a3, 28(sp) + \\ sw a4, 32(sp) + \\ sw a5, 36(sp) + \\ sw a6, 40(sp) + \\ sw a7, 44(sp) + \\ sw t3, 48(sp) + \\ sw t4, 52(sp) + \\ sw t5, 56(sp) + \\ sw t6, 60(sp) + \\ csrr a0, mcause + // Bit 31 set means interrupt, so mcause read as *signed* is negative. `bgez` therefore + // branches exactly on "this was an exception", in one instruction and with no scratch + // register - which matters here because every scratch register is already spoken for. + \\ bgez a0, 1f + // mcause[11:0] is the CLIC's interrupt ID. Isolated with a shift pair rather than `andi`: + // andi's immediate is 12-bit *signed*, so `andi a0, a0, 0xfff` does not assemble as a + // 12-bit mask - it is -1, and would leave the interrupt bit and the level field in place. + \\ slli a0, a0, 20 + \\ srli a0, a0, 20 + \\ call intrDispatch + \\ lw ra, 0(sp) + \\ lw t0, 4(sp) + \\ lw t1, 8(sp) + \\ lw t2, 12(sp) + \\ lw a0, 16(sp) + \\ lw a1, 20(sp) + \\ lw a2, 24(sp) + \\ lw a3, 28(sp) + \\ lw a4, 32(sp) + \\ lw a5, 36(sp) + \\ lw a6, 40(sp) + \\ lw a7, 44(sp) + \\ lw t3, 48(sp) + \\ lw t4, 52(sp) + \\ lw t5, 56(sp) + \\ lw t6, 60(sp) + \\ addi sp, sp, 64 + \\ mret + // The exception arm. No restore and no `mret`: `intrFault` is noreturn, because resuming + // would re-execute the faulting instruction. The saved registers stay on the stack, which + // costs 64 bytes that are never reclaimed and is the correct trade for a path that ends in + // a parked core with the numbers printed. + \\1: + \\ csrr a1, mepc + \\ csrr a2, mtval + \\ call intrFault + ); +} + +// ------------------------------------------------------------------------------------ tests + +test "the enum's IDs are the offsets of the matrix registers they name" { + // The whole of `routeId` rests on `map_reg_addr == base + 4*id`. These four are checked against + // the addresses ESP-IDF's own interrupt_core0_reg.h computes, which is an independent path: + // IDF wrote the offset as a literal per source, this file multiplies. + try std.testing.expectEqual(@as(u32, 0x7c), 4 * @as(u32, @intFromEnum(Source.uart0))); + try std.testing.expectEqual(@as(u32, 0xb0), 4 * @as(u32, @intFromEnum(Source.i2c0))); + try std.testing.expectEqual(@as(u32, 0xd0), 4 * @as(u32, @intFromEnum(Source.ledc))); + try std.testing.expectEqual(@as(u32, 0x1fc), 4 * @as(u32, @intFromEnum(Source.assist_debug))); +} + +test "rev-3-only sources are flagged and the pre-v3 ones are not" { + try std.testing.expect(Source.axi_perf_mon.isRev3Only()); + try std.testing.expect(Source.dma2d_in_ch2.isRev3Only()); + try std.testing.expect(!Source.assist_debug.isRev3Only()); + try std.testing.expect(!Source.dma2d_in_ch1.isRev3Only()); +} + +test "priority and threshold use different encodings of the same three bits" { + // Priority pads low with zeros, threshold pads low with ones. Getting these the same way round + // is the mistake this test exists to catch. + const priority_byte = @as(u32, 5) << nlbits_shift; + const threshold_byte = (@as(u32, 5) << nlbits_shift) | nlbits_pad; + try std.testing.expectEqual(@as(u32, 0xa0), priority_byte); + try std.testing.expectEqual(@as(u32, 0xbf), threshold_byte); + try std.testing.expectEqual(@as(u32, 5), priority_byte >> nlbits_shift); + try std.testing.expectEqual(@as(u32, 5), threshold_byte >> nlbits_shift); +} + +test "trigger's low bit, not its value, decides edge versus level" { + try std.testing.expect(Trigger.rising_edge.isEdge()); + try std.testing.expect(Trigger.falling_edge.isEdge()); + try std.testing.expect(!Trigger.level.isEdge()); + try std.testing.expect(!Trigger.level_alias.isEdge()); +} + +test "the vector table is aligned to a power of two above its own size" { + try std.testing.expectEqual(@as(usize, 256), @alignOf(@TypeOf(vector_table))); + try std.testing.expect(@sizeOf(@TypeOf(vector_table)) <= 256); +} + +test "mcause's sign bit is what separates an interrupt from an exception" { + // The trap entry branches on `bgez mcause`, which is only correct if bit 31 is the interrupt + // flag and the value is read signed. Spelled out here because the asm cannot say it. + const interrupt_mcause: u32 = 0x8000_0015; // CLIC ID 21 = external line 5 + const exception_mcause: u32 = 0x0000_0002; // illegal instruction + try std.testing.expect(@as(i32, @bitCast(interrupt_mcause)) < 0); + try std.testing.expect(@as(i32, @bitCast(exception_mcause)) >= 0); + // And the ID extraction the two shifts perform. + try std.testing.expectEqual(@as(u32, 21), (interrupt_mcause << 20) >> 20); +} + +test "mtvec's mode bits do not collide with a 64-byte-aligned base" { + // The hardware target is `mtvec[31:6] << 6`, so the mode goes in bits the base cannot use - + // but only if the base really is 64-byte aligned. This is the arithmetic `init` performs; + // whether the *linked* trapEntry satisfies it is a fact about the link, and + // `trapEntryAddress()` is how a test on the die checks that, the image being stripped. + const aligned_base: u32 = 0x4000_1200; + const mtvec = aligned_base | mtvec_mode_clic; + try std.testing.expectEqual(aligned_base, (mtvec >> 6) << 6); + // A base one instruction short of alignment vectors 60 bytes early, silently. + const bad_base: u32 = 0x4000_1204; + try std.testing.expect(((bad_base | mtvec_mode_clic) >> 6) << 6 != bad_base); +} diff --git a/src/hal/ledc.zig b/src/hal/ledc.zig new file mode 100644 index 0000000..62eacd8 --- /dev/null +++ b/src/hal/ledc.zig @@ -0,0 +1,587 @@ +//! LEDC: the LED PWM controller. Four timers, eight channels, and the first **shadow-register** +//! peripheral in this HAL. +//! +//! Three things make LEDC different from everything else here, and all three are load-bearing. +//! +//! **1. Configuration is staged, then committed.** `LEDC_PARA_UP_CHn` (channel) and +//! `LEDC_TIMERn_PARA_UP` (timer) are write-to-trigger bits: writing 1 copies the staged fields into +//! the shadow registers the counter and comparators actually use, and the hardware clears the bit +//! again by itself (`ledc_reg.h:42-47`, `:951-958`). Values written without a commit are visible in +//! the register file and have no effect on the output. So every mutator here stages, and every +//! commit is its own store - `commitChannel` / `commitTimer` - exactly as ESP-IDF's +//! `ledc_ll_ls_channel_update` (ledc_ll.h:435-438) and `ledc_ll_ls_timer_update` (ledc_ll.h:286-290) +//! do it. +//! +//! The commit store is a read-modify-write, and that is deliberate rather than sloppy: the commit +//! bit shares its word with the staged fields it commits. `LEDC_PARA_UP_CH0` is bit 4 of +//! `LEDC_CH0_CONF0_REG`, whose other fields are `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV` and `OVF_*`, so +//! a bare `writeRaw(1 << 4)` would erase the very configuration it was meant to commit. Compare +//! `systimer.zig`'s `op.write(.{update.is(1)})`, which is a single whole-word store because +//! `SYSTIMER_UNIT0_OP_REG` contains nothing else. The read-modify-write is safe here for the reason +//! `mmio.zig` gives: `PARA_UP` is `WT`, it reads back 0, so the read half of the read-modify-write +//! can never re-trigger an earlier commit. That is the difference between a self-clearing bit and a +//! write-1-to-clear bit, and it is why LEDC does not need the interrupt-status treatment. +//! +//! **2. The divider is fixed point, Q10.8.** `LEDC_CLK_DIV_TIMERn` is an 18-bit field at [22:5] +//! (`ledc_reg.h:921-928`) holding a divider with 8 fractional bits (`LEDC_LL_FRACTIONAL_BITS`, +//! ledc_ll.h:30): bits [17:8] are the integer part, bits [7:0] the fraction, so the value 0x4E2 +//! means 1250/256 = 4.8828. The output frequency is +//! +//! f_pwm = f_src * 256 / (div * 2^duty_res) +//! +//! and `divisor()` below is ESP-IDF's arithmetic for the inverse, transcribed operation for +//! operation from `esp_driver_ledc/src/ledc.c:468-497` - including the two places where it is +//! surprising. See its comment. +//! +//! **3. On the P4 the clock mux left the peripheral.** `LEDC_CONF_REG.LEDC_APB_CLK_SEL` still exists +//! in the register map and still documents an encoding (0: APB, 1: RC_FAST, 2: XTAL), and ESP-IDF's +//! P4 LL never touches it: the real mux is `HP_SYS_CLKRST.PERI_CLK_CTRL22.REG_LEDC_CLK_SRC_SEL`, +//! with a *different* encoding (0: XTAL, 1: RC_FAST, 2: PLL_DIV) - ledc_ll.h:223-242. Writing the +//! in-block register would silently do nothing, and reading it back to check would silently agree. +//! `ClockSource` below is the HP_SYS_CLKRST encoding. +//! +//! Gamma fade *ramps* are out of scope, but one gamma register is not optional: the P4 moved +//! `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` out of `LEDC_CHn_CONF1_REG` - which on this die +//! holds only `DUTY_START` - and into gamma RAM. A constant duty is therefore a degenerate one-step +//! fade, and `setDuty` writes that single entry, which is what ESP-IDF's `ledc_duty_config` does for +//! every plain duty change (ledc.c:263-280). + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); +const gpio = @import("gpio.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// Eight channels, four timers (`soc_caps.h:385-386`). +pub const channel_count = 8; +pub const timer_count = 4; + +/// The counter is 20 bits, so the duty resolution is at most 20 (`soc_caps.h:387`). The register +/// field is five bits wide and will happily accept 21-31; the hardware will not. +pub const max_duty_resolution = 20; + +/// Fractional bits in `LEDC_CLK_DIV_TIMERn` - `LEDC_LL_FRACTIONAL_BITS`, ledc_ll.h:30. +pub const fractional_bits = 8; + +/// The divider must be at least 1.0 and must fit the field: ESP-IDF's `LEDC_IS_DIV_INVALID` +/// (ledc.c:114) rejects anything `<= LEDC_LL_FRACTIONAL_MAX` or `> LEDC_TIMER_DIV_NUM_MAX`. +pub const divisor_min: u32 = 1 << fractional_bits; +pub const divisor_max: u32 = 0x3ffff; + +pub const Error = error{ + /// The requested frequency cannot be reached from this source at this resolution: the divider + /// would be below 1.0 (frequency too high) or wider than 18 bits (frequency too low). + DividerOutOfRange, + DutyResolutionOutOfRange, +}; + +// ------------------------------------------------------------------------------------- registers + +// Five registers per channel, stride 0x14; two per timer, stride 0x08. Both strides come from a +// second instance's macro rather than being assumed - see mmio.RegArray. +const ch_conf0 = mmio.RegArray(regs.LEDC_CH0_CONF0_REG, regs.LEDC_CH1_CONF0_REG, channel_count); +const ch_hpoint = mmio.RegArray(regs.LEDC_CH0_HPOINT_REG, regs.LEDC_CH1_HPOINT_REG, channel_count); +const ch_duty = mmio.RegArray(regs.LEDC_CH0_DUTY_REG, regs.LEDC_CH1_DUTY_REG, channel_count); +const ch_conf1 = mmio.RegArray(regs.LEDC_CH0_CONF1_REG, regs.LEDC_CH1_CONF1_REG, channel_count); +const ch_duty_r = mmio.RegArray(regs.LEDC_CH0_DUTY_R_REG, regs.LEDC_CH1_DUTY_R_REG, channel_count); +const ch_gamma_conf = mmio.RegArray(regs.LEDC_CH0_GAMMA_CONF_REG, regs.LEDC_CH1_GAMMA_CONF_REG, channel_count); +// Gamma RAM: 16 entries per channel, so the per-channel stride is 0x40 and entry 0 is the base. +const ch_gamma_range0 = mmio.RegArray(regs.LEDC_CH0_GAMMA_RANGE0_REG, regs.LEDC_CH1_GAMMA_RANGE0_REG, channel_count); +const tim_conf = mmio.RegArray(regs.LEDC_TIMER0_CONF_REG, regs.LEDC_TIMER1_CONF_REG, timer_count); +const tim_value = mmio.RegArray(regs.LEDC_TIMER0_VALUE_REG, regs.LEDC_TIMER1_VALUE_REG, timer_count); + +// Field geometry is taken from instance 0 and reused for every instance, which is only sound if the +// instances agree; the comptime block below checks the ends of both ranges against instance 0. That +// is not paranoia about the silicon, it is paranoia about the macro names: `LEDC_CLK_DIV_TIMER0` and +// `LEDC_TIMER0_DUTY_RES` put the instance number in different places, and picking up +// `LEDC_TIMER1_DUTY_RES_S` while meaning timer 0's shift is a one-character mistake. +const timer_sel = Field.of(regs.LEDC_TIMER_SEL_CH0_S, regs.LEDC_TIMER_SEL_CH0_V); +const sig_out_en = Field.of(regs.LEDC_SIG_OUT_EN_CH0_S, regs.LEDC_SIG_OUT_EN_CH0_V); +const idle_lv = Field.of(regs.LEDC_IDLE_LV_CH0_S, regs.LEDC_IDLE_LV_CH0_V); +const ch_para_up = Field.of(regs.LEDC_PARA_UP_CH0_S, regs.LEDC_PARA_UP_CH0_V); +const hpoint = Field.of(regs.LEDC_HPOINT_CH0_S, regs.LEDC_HPOINT_CH0_V); +const duty = Field.of(regs.LEDC_DUTY_CH0_S, regs.LEDC_DUTY_CH0_V); +const duty_r = Field.of(regs.LEDC_DUTY_CH0_R_S, regs.LEDC_DUTY_CH0_R_V); +const duty_start = Field.of(regs.LEDC_DUTY_START_CH0_S, regs.LEDC_DUTY_START_CH0_V); +const gamma_entry_num = Field.of(regs.LEDC_CH0_GAMMA_ENTRY_NUM_S, regs.LEDC_CH0_GAMMA_ENTRY_NUM_V); +const gamma_duty_inc = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_INC_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_INC_V); +const gamma_duty_cycle = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_CYCLE_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_CYCLE_V); +const gamma_scale = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_SCALE_S, regs.LEDC_CH0_GAMMA_RANGE0_SCALE_V); +const gamma_duty_num = Field.of(regs.LEDC_CH0_GAMMA_RANGE0_DUTY_NUM_S, regs.LEDC_CH0_GAMMA_RANGE0_DUTY_NUM_V); + +const duty_res = Field.of(regs.LEDC_TIMER0_DUTY_RES_S, regs.LEDC_TIMER0_DUTY_RES_V); +const clk_div = Field.of(regs.LEDC_CLK_DIV_TIMER0_S, regs.LEDC_CLK_DIV_TIMER0_V); +const tim_pause = Field.of(regs.LEDC_TIMER0_PAUSE_S, regs.LEDC_TIMER0_PAUSE_V); +const tim_rst = Field.of(regs.LEDC_TIMER0_RST_S, regs.LEDC_TIMER0_RST_V); +const tim_para_up = Field.of(regs.LEDC_TIMER0_PARA_UP_S, regs.LEDC_TIMER0_PARA_UP_V); + +comptime { + const same = struct { + fn check(comptime what: []const u8, comptime a: Field, comptime b: Field) void { + if (a.shift != b.shift or a.width != b.width) @compileError( + "the per-instance " ++ what ++ " macros disagree on bit position or width; " ++ + "this file must index the field per instance instead of reusing instance 0's", + ); + } + }.check; + // Channels: 1 and 7, the two ends of the range beyond instance 0. + same("LEDC_TIMER_SEL_CHn", timer_sel, Field.of(regs.LEDC_TIMER_SEL_CH1_S, regs.LEDC_TIMER_SEL_CH1_V)); + same("LEDC_TIMER_SEL_CHn", timer_sel, Field.of(regs.LEDC_TIMER_SEL_CH7_S, regs.LEDC_TIMER_SEL_CH7_V)); + same("LEDC_SIG_OUT_EN_CHn", sig_out_en, Field.of(regs.LEDC_SIG_OUT_EN_CH7_S, regs.LEDC_SIG_OUT_EN_CH7_V)); + same("LEDC_IDLE_LV_CHn", idle_lv, Field.of(regs.LEDC_IDLE_LV_CH7_S, regs.LEDC_IDLE_LV_CH7_V)); + same("LEDC_PARA_UP_CHn", ch_para_up, Field.of(regs.LEDC_PARA_UP_CH7_S, regs.LEDC_PARA_UP_CH7_V)); + same("LEDC_HPOINT_CHn", hpoint, Field.of(regs.LEDC_HPOINT_CH7_S, regs.LEDC_HPOINT_CH7_V)); + same("LEDC_DUTY_CHn", duty, Field.of(regs.LEDC_DUTY_CH7_S, regs.LEDC_DUTY_CH7_V)); + same("LEDC_DUTY_START_CHn", duty_start, Field.of(regs.LEDC_DUTY_START_CH7_S, regs.LEDC_DUTY_START_CH7_V)); + same("LEDC_CHn_GAMMA_ENTRY_NUM", gamma_entry_num, Field.of(regs.LEDC_CH7_GAMMA_ENTRY_NUM_S, regs.LEDC_CH7_GAMMA_ENTRY_NUM_V)); + same("LEDC_CHn_GAMMA_RANGE0_SCALE", gamma_scale, Field.of(regs.LEDC_CH7_GAMMA_RANGE0_SCALE_S, regs.LEDC_CH7_GAMMA_RANGE0_SCALE_V)); + // Timers: 1 and 3. + same("LEDC_TIMERn_DUTY_RES", duty_res, Field.of(regs.LEDC_TIMER1_DUTY_RES_S, regs.LEDC_TIMER1_DUTY_RES_V)); + same("LEDC_TIMERn_DUTY_RES", duty_res, Field.of(regs.LEDC_TIMER3_DUTY_RES_S, regs.LEDC_TIMER3_DUTY_RES_V)); + same("LEDC_CLK_DIV_TIMERn", clk_div, Field.of(regs.LEDC_CLK_DIV_TIMER3_S, regs.LEDC_CLK_DIV_TIMER3_V)); + same("LEDC_TIMERn_PAUSE", tim_pause, Field.of(regs.LEDC_TIMER3_PAUSE_S, regs.LEDC_TIMER3_PAUSE_V)); + same("LEDC_TIMERn_RST", tim_rst, Field.of(regs.LEDC_TIMER3_RST_S, regs.LEDC_TIMER3_RST_V)); + same("LEDC_TIMERn_PARA_UP", tim_para_up, Field.of(regs.LEDC_TIMER3_PARA_UP_S, regs.LEDC_TIMER3_PARA_UP_V)); + + // `LEDC_TIMER_DIV_NUM_MAX` (ledc.c:110) is a literal in the driver; it should be the field's + // own mask, and if a future die widens the field this is where the two part company. + if (divisor_max != clk_div.max()) @compileError( + "divisor_max no longer matches LEDC_CLK_DIV_TIMERn's width", + ); + // The eight output signals must be consecutive for `signalIndex` to be arithmetic. + if (regs.LEDC_LS_SIG_OUT_PAD_OUT7_IDX - regs.LEDC_LS_SIG_OUT_PAD_OUT0_IDX != channel_count - 1) + @compileError("the LEDC output signal indices are not consecutive; signalIndex must be a table"); +} + +// ------------------------------------------------------------------------------ clocks and reset + +/// LEDC's function clock, in HP_SYS_CLKRST rather than in the peripheral (ledc_ll.h:179, :241). +/// Shared with RMT's fields, hence the interrupt-masked read-modify-write. +const peri_clk_ctrl22 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL22_REG); +const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_LEDC_CLK_SRC_SEL_V); +const func_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_LEDC_CLK_EN_S, regs.HP_SYS_CLKRST_REG_LEDC_CLK_EN_V); + +/// The four timers' shared source. Encoding from `ledc_ll_set_slow_clk_sel` (ledc_ll.h:223-242) - +/// *not* the encoding `LEDC_CONF_REG.APB_CLK_SEL` documents, which is a different register on a +/// different block and is dead on this die. +pub const ClockSource = enum(u2) { + /// 40 MHz on this board (`clk_tree_defs.h:145`). + xtal = 0, + /// The internal RC oscillator: approximately 17.5 MHz (`clk_tree_defs.h:58`) and not trimmed. + /// ESP-IDF calibrates it against XTAL before using it for a divider; there is no calibration + /// here, so a frequency computed from `rc_fast_hz_approx` is approximate too. + rc_fast = 1, + /// PLL_F80M, 80 MHz (`clk_tree_defs.h:168`). Called `LEDC_SLOW_CLK_PLL_DIV` by ESP-IDF. + pll_div = 2, + + /// The source frequency to feed `divisor`, or null for RC_FAST, whose real rate has to be + /// measured rather than assumed. + pub fn hz(self: ClockSource) ?u32 { + return switch (self) { + .xtal => xtal_hz, + .pll_div => pll_div_hz, + .rc_fast => null, + }; + } +}; + +pub const xtal_hz: u32 = 40_000_000; +pub const pll_div_hz: u32 = 80_000_000; +pub const rc_fast_hz_approx: u32 = 17_500_000; + +/// Select the timers' source clock. A read-modify-write of a register that also holds RMT's clock +/// fields, so it runs with interrupts masked, like everything else that touches HP_SYS_CLKRST. +pub fn setClockSource(src: ClockSource) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl22.modify(.{clk_src_sel.is(@intFromEnum(src))}); +} + +pub fn getClockSource() ClockSource { + return @enumFromInt(peri_clk_ctrl22.get(clk_src_sel)); +} + +/// LEDC's core ("function") clock gate. Distinct from the APB gate in `clkrst`: the APB clock makes +/// the registers addressable, this one makes the counters run - and ESP-IDF notes that some LEDC +/// registers and the gamma RAM need it just to be read or written (ledc.c:433-436). +pub fn setFunctionClockEnabled(on: bool) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + peri_clk_ctrl22.modify(.{func_clk_en.is(@intFromBool(on))}); +} + +/// Bring the peripheral up, in the only order that works: bus clock, reset, function clock, source. +/// +/// The bus clock first because LEDC is one of the blocks whose APB gate is *off* at power-on +/// (`hp_sys_clkrst_reg.h:835`, REG_LEDC_APB_CLK_EN default 0), so every register read before this +/// returns the last value the bus latched. The function clock before any configuration because the +/// gamma RAM needs it. ESP-IDF deasserts the reset rather than pulsing it (ledc.c:430-431), because +/// its driver may be attaching to a running LEDC; this pulses, which is the stronger guarantee for a +/// fresh boot and is measurably safe on this board - pulsing REG_RST_EN_LEDC for 1 ms left the +/// console untouched and returned LEDC_CH0_CONF0 to 0. +pub fn init(src: ClockSource) void { + clkrst.setClockEnabled(.ledc, true); + clkrst.resetPeripheral(.ledc); + setFunctionClockEnabled(true); + setClockSource(src); +} + +// -------------------------------------------------------------------------------- divider maths + +/// ESP-IDF's `ledc_calculate_divisor`, transcribed from `esp_driver_ledc/src/ledc.c:468-497`: +/// +/// return (((uint64_t) src_clk_freq << LEDC_LL_FRACTIONAL_BITS) + freq_hz * precision / 2) +/// / (freq_hz * precision); +/// +/// Result is Q10.8 - see the file comment - and `divisorValid` says whether it fits the field. +/// +/// Two properties of that C expression are not obvious and are reproduced deliberately, because a +/// HAL that computed a *better* divider than IDF's would disagree with it on real inputs and there +/// would be no way to tell which of the two was wrong: +/// +/// 1. `freq_hz * precision` is `int * uint32_t`, so it is computed in **32 bits and wraps**, and +/// the wrap is not always harmlessly out of range. Ask for 4097 Hz at 20-bit resolution from the +/// 40 MHz XTAL: the true product is 2^32 + 2^20, the C code divides by 2^20 instead, and the +/// answer is 9766 - a *valid* divider, which programs 1.0 Hz. IDF accepts it, because the value +/// passes its own range check. `%*` here is that wrap, on purpose: reproducing it is what makes +/// the on-die comparison meaningful, and the numbers above are how a caller can recognise it. +/// 2. The quotient is `uint64_t` but the return type is `uint32_t`, so it is **truncated**. From a +/// 40 MHz source at 1 Hz and 1-bit resolution the quotient is 5.12e9 and IDF returns 825032704. +/// `@truncate` is that truncation. +/// +/// The one place this cannot follow IDF is `freq_hz * precision == 0`, reachable at exactly 4096 Hz +/// with 20-bit resolution (2^32, wrapping to zero), where the C code divides by zero. Returning 0 is +/// a deliberate substitution: it is not a valid divider, so `divisorValid` rejects it and the caller +/// gets an error instead of undefined behaviour. +pub fn divisor(src_hz: u32, freq_hz: u32, resolution: u5) u32 { + const precision: u32 = @as(u32, 1) << resolution; + const den: u32 = freq_hz *% precision; + if (den == 0) return 0; + const num: u64 = (@as(u64, src_hz) << fractional_bits) + den / 2; + return @truncate(num / den); +} + +/// `LEDC_IS_DIV_INVALID`, inverted (ledc.c:114). A divider below 1.0 means the requested frequency +/// is faster than the source can produce at that resolution. +pub fn divisorValid(div: u32) bool { + return div >= divisor_min and div <= divisor_max; +} + +/// The frequency a given divider and resolution actually produce: `f_src * 256 / (div * 2^res)`, +/// rounded, and 0 for a divider of 0. +/// +/// This is `ledc_get_freq`'s arithmetic (ledc.c:1175) with one deliberate difference: the +/// denominator is computed in 64 bits, so it does not wrap. Nothing compares this against IDF - it +/// is a convenience for callers checking what they got - and a wrapped denominator here would be a +/// bug rather than a compatibility requirement. +pub fn frequencyOf(src_hz: u32, div: u32, resolution: u5) u32 { + if (div == 0) return 0; + const den: u64 = @as(u64, div) * (@as(u64, 1) << resolution); + const num: u64 = (@as(u64, src_hz) << fractional_bits) + den / 2; + return @truncate(num / den); +} + +// --------------------------------------------------------------------------------------- timers + +/// Stage the divider. `ledc_ll_set_clock_divider`, ledc_ll.h:345-348. +pub fn setClockDivider(timer: u32, div: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{clk_div.is(div)}); +} + +pub fn getClockDivider(timer: u32) u32 { + std.debug.assert(timer < timer_count); + return tim_conf.at(timer).get(clk_div); +} + +/// Stage the duty resolution, in bits. `ledc_ll_set_duty_resolution`, ledc_ll.h:391-394. +pub fn setDutyResolution(timer: u32, bits: u5) void { + std.debug.assert(timer < timer_count); + std.debug.assert(bits <= max_duty_resolution); + tim_conf.at(timer).modify(.{duty_res.is(bits)}); +} + +pub fn getDutyResolution(timer: u32) u5 { + std.debug.assert(timer < timer_count); + return @intCast(tim_conf.at(timer).get(duty_res)); +} + +/// Commit the staged divider and resolution. One store, and the bit clears itself. +/// +/// ESP-IDF does not wait for it: "we don't wait for the bit gets cleared since it can take quite +/// long depends on the pwm frequency" (ledc_ll.h:289). Neither does this - a poll here would block +/// for a whole PWM period, and there is nothing useful to do with the answer. +pub fn commitTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{tim_para_up.is(1)}); +} + +/// Reset the timer's counter: assert, deassert (`ledc_ll_timer_rst`, ledc_ll.h:301-305). +/// +/// Note the reset value of `LEDC_TIMERn_RST` is **1** (ledc_reg.h:936-943), which is one of the +/// 46.7% of fields whose reset value is not zero, and the reason `configureTimer` finishes by +/// clearing it: a freshly reset LEDC block holds all four counters at zero and they stay there until +/// something writes that bit back down. +pub fn resetTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + const r = tim_conf.at(timer); + r.modify(.{tim_rst.is(1)}); + r.modify(.{tim_rst.is(0)}); +} + +/// Freeze the counter where it is (`ledc_ll_timer_pause`, ledc_ll.h:316-319). +pub fn pauseTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{tim_pause.is(1)}); +} + +pub fn resumeTimer(timer: u32) void { + std.debug.assert(timer < timer_count); + tim_conf.at(timer).modify(.{tim_pause.is(0)}); +} + +/// The counter's current value, 20 bits. Reading it is a plain load - no latch handshake, unlike +/// systimer. +pub fn timerCount(timer: u32) u32 { + std.debug.assert(timer < timer_count); + return tim_value.at(timer).raw(); +} + +/// Everything a timer needs to produce `freq_hz` at `resolution` bits, in ESP-IDF's order: +/// divider, resolution, commit, then out of pause and out of reset (`ledc_set_timer_params`, +/// ledc.c:244-261, followed by ledc.c:816-818). +/// +/// Returns `DividerOutOfRange` rather than programming a divider the hardware cannot hold. The +/// caller passes the source frequency because this HAL has no clock tree: `ClockSource.hz()` gives +/// it for XTAL and PLL_DIV, and RC_FAST has to be measured. +pub fn configureTimer(timer: u32, opts: struct { + src_hz: u32, + freq_hz: u32, + resolution: u5, +}) Error!void { + std.debug.assert(timer < timer_count); + if (opts.resolution == 0 or opts.resolution > max_duty_resolution) return Error.DutyResolutionOutOfRange; + const div = divisor(opts.src_hz, opts.freq_hz, opts.resolution); + if (!divisorValid(div)) return Error.DividerOutOfRange; + + setClockDivider(timer, div); + setDutyResolution(timer, opts.resolution); + commitTimer(timer); + resumeTimer(timer); + resetTimer(timer); +} + +// ------------------------------------------------------------------------------------- channels + +/// Which timer drives this channel. Staged; needs `commitChannel`. +/// `ledc_ll_bind_channel_timer`, ledc_ll.h:697-700. +pub fn bindTimer(channel: u32, timer: u32) void { + std.debug.assert(channel < channel_count and timer < timer_count); + ch_conf0.at(channel).modify(.{timer_sel.is(timer)}); +} + +pub fn boundTimer(channel: u32) u32 { + std.debug.assert(channel < channel_count); + return ch_conf0.at(channel).get(timer_sel); +} + +/// Where in the period the output goes high, in counter ticks. Staged. +/// `ledc_ll_set_hpoint`, ledc_ll.h:450-453. +pub fn setHpoint(channel: u32, value: u32) void { + std.debug.assert(channel < channel_count and value <= hpoint.max()); + ch_hpoint.at(channel).modify(.{hpoint.is(value)}); +} + +/// Stage a duty value, in counter ticks out of `2^resolution`. +/// +/// Two things happen here that the name does not suggest, and both are ESP-IDF's +/// (`ledc_ll_set_duty_int_part` ledc_ll.h:480-483, `ledc_duty_config` ledc.c:263-280): +/// +/// * The register holds duty in **Q21.4** - four fractional bits, used by fades - so the integer +/// duty is shifted left by 4. `getDuty` shifts back. +/// * The P4 has no plain-duty path. `DUTY_NUM`/`DUTY_CYCLE`/`DUTY_SCALE`/`DUTY_INC` moved out of +/// `CHn_CONF1` into gamma RAM, so a constant duty is a one-step fade of scale 0: entry 0 gets +/// (increase, one cycle, scale 0, one step) and the range count is set to 1. Without that entry +/// the staged duty is committed and the output does not move. +/// +/// Staged; needs `commitChannel` (or `start`, which commits). +pub fn setDuty(channel: u32, value: u32) void { + std.debug.assert(channel < channel_count); + std.debug.assert(value <= duty.max() >> 4); + ch_duty.at(channel).modify(.{duty.is(value << 4)}); + stageNoFade(channel); +} + +/// The duty the hardware is currently using, from the read-only shadow (`ledc_ll_get_duty`, +/// ledc_ll.h:495-498). This is the one register that shows whether a commit actually happened - and +/// it only updates when the timer next overflows, so it is not a synchronous read-back. +pub fn currentDuty(channel: u32) u32 { + std.debug.assert(channel < channel_count); + return ch_duty_r.at(channel).get(duty_r) >> 4; +} + +/// Gamma RAM entry 0 as "no fade": one step, one cycle, scale 0, increasing. Exactly the parameters +/// `ledc_set_duty` passes down (ledc.c:1109-1117) for a constant duty. +fn stageNoFade(channel: u32) void { + // The whole word is being established, and every field in it is being named, so this is one of + // the few places `write` is right rather than `modify`. + ch_gamma_range0.at(channel).write(.{ + gamma_duty_inc.is(1), + gamma_duty_cycle.is(1), + gamma_scale.is(0), + gamma_duty_num.is(1), + }); + ch_gamma_conf.at(channel).modify(.{gamma_entry_num.is(1)}); +} + +/// The output driver. Staged; needs `commitChannel`. +/// `ledc_ll_set_sig_out_en`, ledc_ll.h:592-596. +pub fn setOutputEnabled(channel: u32, on: bool) void { + std.debug.assert(channel < channel_count); + ch_conf0.at(channel).modify(.{sig_out_en.is(@intFromBool(on))}); +} + +/// The level the pad holds while the channel is disabled - and only while it is disabled +/// (`ledc_reg.h:34-37`: "Valid only when LEDC_SIG_OUT_EN_CHn is 0"). Staged. +/// `ledc_ll_set_idle_level`, ledc_ll.h:622-626. +pub fn setIdleLevel(channel: u32, level: u1) void { + std.debug.assert(channel < channel_count); + ch_conf0.at(channel).modify(.{idle_lv.is(level)}); +} + +/// Hand the staged duty to the fade engine. `ledc_ll_set_duty_start`, ledc_ll.h:607-610. +/// +/// `DUTY_START` lives in `CHn_CONF1`, alone, and is annotated `R/W/SC` - the hardware clears it when +/// the (here one-step) fade finishes. A read-modify-write is still the right store: the bit is the +/// only field in the word, but bits 30:0 are reserved and writing them back as read is what IDF's +/// bitfield assignment does. +pub fn startFade(channel: u32) void { + std.debug.assert(channel < channel_count); + ch_conf1.at(channel).modify(.{duty_start.is(1)}); +} + +/// Commit the channel's staged fields: `TIMER_SEL`, `SIG_OUT_EN`, `IDLE_LV`, `HPOINT`, +/// `DUTY_START`, `OVF_CNT_EN` and the duty (`ledc_reg.h:42-47`). +/// +/// One deliberate store, never folded into the store that staged the values, matching +/// `ledc_ll_ls_channel_update` (ledc_ll.h:435-438). It is a read-modify-write because the commit bit +/// shares its word with the staged fields - see the file comment - and that is safe only because the +/// bit reads back as 0. +pub fn commitChannel(channel: u32) void { + std.debug.assert(channel < channel_count); + ch_conf0.at(channel).modify(.{ch_para_up.is(1)}); +} + +/// Start driving: output on, duty handed over, committed. `_ledc_update_duty`, ledc.c:1021-1026. +pub fn start(channel: u32) void { + setOutputEnabled(channel, true); + startFade(channel); + commitChannel(channel); +} + +/// Stop driving and hold the pad at `idle_level`. `ledc_stop`, ledc.c:1039-1050. +/// +/// The order is IDF's and it matters: the idle level is staged *before* the output is disabled, so +/// the two reach the hardware in the same commit and the pad never spends a period at the old idle +/// level. +pub fn stop(channel: u32, idle_level: u1) void { + setIdleLevel(channel, idle_level); + setOutputEnabled(channel, false); + commitChannel(channel); +} + +/// A whole channel in one commit: timer, duty, hpoint, idle level, output enable. +/// +/// This is the one operation here that is not a transcription of an ESP-IDF function - IDF's +/// `ledc_channel_config` also allocates a driver object, reserves the pin and installs a fade +/// service - but it is the same register sequence: stage everything, then commit once. One commit +/// rather than five is the point: the channel changes all at once, at a period boundary, instead of +/// drifting through four intermediate configurations. +pub fn configureChannel(channel: u32, opts: struct { + timer: u32, + duty: u32, + hpoint: u32 = 0, + idle_level: u1 = 0, + output_enabled: bool = true, +}) void { + bindTimer(channel, opts.timer); + setHpoint(channel, opts.hpoint); + setDuty(channel, opts.duty); + setIdleLevel(channel, opts.idle_level); + setOutputEnabled(channel, opts.output_enabled); + startFade(channel); + commitChannel(channel); +} + +// ------------------------------------------------------------------------------------ pin output + +/// The GPIO matrix signal index for a channel's output. `ledc_periph_signal[0].sig_out0_idx` is +/// `LEDC_LS_SIG_OUT_PAD_OUT0_IDX` (esp_hal_ledc/esp32p4/ledc_periph.c:14-18) and the driver adds the +/// channel number to it (ledc.c:831); the eight indices are consecutive from 126, asserted above. +pub fn signalIndex(channel: u32) u32 { + std.debug.assert(channel < channel_count); + return @as(u32, @intCast(regs.LEDC_LS_SIG_OUT_PAD_OUT0_IDX)) + channel; +} + +/// Route a channel's output to a pad through the GPIO matrix. No LEDC register is involved: the +/// peripheral has no pad of its own, and this is the whole of `ledc_set_pin`'s hardware effect +/// (ledc.c:823-836, whose `gpio_matrix_output` is func_sel + matrix source + output-enable control, +/// gpio_hal.c:60-69). +pub fn attachPin(channel: u32, pin: u8) void { + gpio.matrixOut(pin, signalIndex(channel)); +} + +// ----------------------------------------------------------------------------------------- tests + +test "the divider is Q10.8: integer part in [17:8], fraction in [7:0]" { + // 40 MHz XTAL, 1 kHz, 13-bit resolution. 40e6*256/(1000*8192) = 1250 = 0x4E2, i.e. 4 + 226/256 + // = 4.8828. Checked against ESP-IDF's own expression compiled on the host over a 1,680-point + // sweep of (source, frequency, resolution). + try std.testing.expectEqual(@as(u32, 1250), divisor(40_000_000, 1_000, 13)); + try std.testing.expectEqual(@as(u32, 1250 >> 8), 4); + try std.testing.expectEqual(@as(u32, 1250 & 0xff), 226); + // And back again, to within the rounding the format allows. + try std.testing.expectEqual(@as(u32, 1_000), frequencyOf(40_000_000, 1250, 13)); +} + +test "divider values for the frequencies the differential harness uses" { + try std.testing.expectEqual(@as(u32, 2000), divisor(40_000_000, 5_000, 10)); + try std.testing.expectEqual(@as(u32, 500), divisor(40_000_000, 20_000, 10)); + try std.testing.expectEqual(@as(u32, 2083), divisor(40_000_000, 300, 14)); + // 80 MHz PLL_F80M, same request: exactly twice the divider. + try std.testing.expectEqual(@as(u32, 4000), divisor(80_000_000, 5_000, 10)); +} + +test "the arithmetic reproduces IDF's overflow and truncation rather than fixing them" { + // 32-bit wrap of freq*precision: the true product at 1 MHz / 13 bits is 8_192_000_000, and the + // C expression divides by 3_897_032_704 instead, giving 3 where the unwrapped arithmetic would + // give 1. Neither is a usable divider - both are below 1.0, so `divisorValid` rejects them the + // way `LEDC_IS_DIV_INVALID` does - but the *value* has to be IDF's, or a caller comparing the + // two implementations sees a difference that is really just two different roundings. + try std.testing.expectEqual(@as(u32, 1_000_000 *% (@as(u32, 1) << 13)), 3_897_032_704); + try std.testing.expectEqual(@as(u32, 3), divisor(40_000_000, 1_000_000, 13)); + try std.testing.expect(!divisorValid(divisor(40_000_000, 1_000_000, 13))); + // u64 quotient truncated to u32, exactly as the C return type does. + try std.testing.expectEqual(@as(u32, 825_032_704), divisor(40_000_000, 1, 1)); + // The one input where IDF divides by zero: 4096 * 2^20 == 2^32. + try std.testing.expectEqual(@as(u32, 0), divisor(40_000_000, 4096, 20)); + try std.testing.expect(!divisorValid(divisor(40_000_000, 4096, 20))); + // The wrap that is *not* self-limiting: 4097 Hz at 20 bits gives a divider IDF's own range check + // accepts, and it programs 1.0 Hz. Reproduced rather than corrected, because the point of the + // differential test is to be wrong in the same way IDF is or not at all. + try std.testing.expectEqual(@as(u32, 9766), divisor(40_000_000, 4097, 20)); + try std.testing.expect(divisorValid(9766)); + try std.testing.expectEqual(@as(u32, 1), frequencyOf(40_000_000, 9766, 20)); +} + +test "validity is the field's range, not the whole u32" { + try std.testing.expect(!divisorValid(0xff)); // below 1.0 + try std.testing.expect(divisorValid(0x100)); // exactly 1.0 + try std.testing.expect(divisorValid(0x3ffff)); + try std.testing.expect(!divisorValid(0x40000)); + // 40 MHz cannot make 5 kHz at 13 bits: that needs a divider of 0.98. + try std.testing.expect(!divisorValid(divisor(40_000_000, 5_000, 13))); +} diff --git a/src/hal/rwdt.zig b/src/hal/rwdt.zig new file mode 100644 index 0000000..5002400 --- /dev/null +++ b/src/hal/rwdt.zig @@ -0,0 +1,123 @@ +//! The RTC watchdog (RWDT) and the super watchdog (SWD), in the always-on LP domain. +//! +//! This module exists because of a bug that had been in every application in this project since the +//! first one, and was invisible for a simple reason: nothing had ever run for more than eight +//! seconds. +//! +//! The second-stage bootloader arms the RTC watchdog to cover the handover to the application, and +//! expects the application to take it over - ESP-IDF disables it in `esp_system`'s startup, which a +//! bare image never runs. So the board resets, and the console says so if anyone looks: +//! +//! MARK ZIG_P4_ALIVE beat=8 gpio20 high=1 low=0 +//! rst:0x10 (CHIP_LP_WDT_RESET),boot:0x30f (SPI_FAST_FLASH_BOOT) +//! +//! Every demo, every example and every hardware test in this repo had been silently rebooting on a +//! roughly ten-second cycle. It surfaced only when the differential harness grew past 26 cases and +//! the run stopped fitting inside one watchdog period - which first looked like "the UART suite +//! crashes the board", and was not. +//! +//! Two watchdogs live here and both have to be dealt with: +//! +//! * **RWDT**, the RTC watchdog proper, in `LP_WDT_CONFIG0_REG`. Write-protected. +//! * **SWD**, the super watchdog, a separate always-on timer whose job is to catch a system that +//! has stopped feeding everything else. It has its own key and its own register, and the +//! bootloader leaves it auto-feeding (`bootloader_super_wdt_auto_feed`); an application that +//! disables RWDT and forgets SWD gets a longer fuse rather than no fuse. +//! +//! The write-protect scheme is the same as the timer groups': the key register's *reset value* is +//! the unlock key, so writing anything else locks it. Writes to a locked register are dropped +//! silently - no fault, no status bit - which is why `disable()` verifies afterwards and returns +//! whether it took. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +const config0 = Reg.at(regs.LP_WDT_CONFIG0_REG); +const wprotect = Reg.at(regs.LP_WDT_WPROTECT_REG); +const swd_config = Reg.at(regs.LP_WDT_SWD_CONFIG_REG); +const swd_wprotect = Reg.at(regs.LP_WDT_SWD_WPROTECT_REG); + +const wdt_en = Field.of(regs.LP_WDT_WDT_EN_S, regs.LP_WDT_WDT_EN_V); +/// Flash-boot mode runs the watchdog independently of `wdt_en`, which is how the bootloader keeps +/// the fuse lit across the handover. Clearing `wdt_en` alone leaves this armed. +const flashboot_en = Field.of(regs.LP_WDT_WDT_FLASHBOOT_MOD_EN_S, regs.LP_WDT_WDT_FLASHBOOT_MOD_EN_V); +const swd_disable = Field.of(regs.LP_WDT_SWD_DISABLE_S, regs.LP_WDT_SWD_DISABLE_V); +const swd_auto_feed = Field.of(regs.LP_WDT_SWD_AUTO_FEED_EN_S, regs.LP_WDT_SWD_AUTO_FEED_EN_V); +const swd_feed = Field.of(regs.LP_WDT_SWD_FEED_S, regs.LP_WDT_SWD_FEED_V); +const feed_reg = Reg.at(regs.LP_WDT_FEED_REG); +const feed_bit = Field.of(regs.LP_WDT_FEED_S, regs.LP_WDT_FEED_V); + +/// The unlock key for both blocks, which is also each key register's reset value: "if the register +/// contains a different value than its reset value, write protection is enabled" +/// (lp_wdt_reg.h). `LP_WDT_WKEY_VALUE` and `LP_WDT_SWD_WKEY_VALUE` in +/// esp_hal_wdt/esp32p4/include/hal/lpwdt_ll.h:25,27 are both this number. +const wkey: u32 = 0x50D8_3AA1; + +/// Unlocked access to the RTC watchdog. `defer guard.release()` re-locks. +pub const Guard = struct { + pub inline fn release(_: Guard) void { + // Anything that is not the key locks it. ESP-IDF writes 0 (lpwdt_ll.h), so this does too: + // it keeps the register comparable against IDF's in a differential test. + wprotect.writeRaw(0); + } +}; + +pub inline fn unlock() Guard { + wprotect.writeRaw(wkey); + return .{}; +} + +/// Turn the RTC watchdog off, and stop the super watchdog behind it. +/// +/// Returns false if the write did not take, which means the key was wrong: a protected register +/// swallows writes without complaint, so the only way to know is to read back. +pub fn disable() bool { + { + const guard = unlock(); + defer guard.release(); + // Both bits, in one store: clearing `wdt_en` while leaving flash-boot mode armed is the + // half-fix that still reboots. + config0.modify(.{ wdt_en.is(0), flashboot_en.is(0) }); + } + + // The super watchdog is a separate block with its own key. + swd_wprotect.writeRaw(wkey); + swd_config.modify(.{ swd_disable.is(1), swd_auto_feed.is(0) }); + swd_wprotect.writeRaw(0); + + return config0.get(wdt_en) == 0 and config0.get(flashboot_en) == 0 and swd_config.get(swd_disable) == 1; +} + +/// Feed the RTC watchdog instead of disabling it, for an application that would rather keep the +/// protection. +/// +/// The counter is fed through its own register, `LP_WDT_FEED_REG`, not through anything in +/// CONFIG0 - ESP-IDF's `lpwdt_ll_feed` writes `hw->feed.feed = 1`. An earlier version of this +/// function read a CONFIG0 field and wrote the same value back, which is a pure no-op: the register +/// ended with the bits it started with and the counter kept running. An application that took this +/// module's own advice - keep the protection, feed it - would have been reset about ten seconds +/// later with nothing on the console, which is the exact failure this file exists to document. The +/// differential harness could not have caught it either, because a no-op leaves the register +/// bit-identical. +pub fn feed() void { + const guard = unlock(); + defer guard.release(); + feed_reg.write(.{feed_bit.is(1)}); +} + +/// Feed the super watchdog once. Independent of RWDT and of its own auto-feed setting. +pub fn feedSuper() void { + swd_wprotect.writeRaw(wkey); + swd_config.modify(.{swd_feed.is(1)}); + swd_wprotect.writeRaw(0); +} + +/// Whether either watchdog is still armed - worth printing once at startup, because the symptom of +/// getting this wrong is a reset ten seconds later with no other clue. +pub fn armed() bool { + return config0.get(wdt_en) == 1 or config0.get(flashboot_en) == 1 or swd_config.get(swd_disable) == 0; +} diff --git a/src/hal/sdmmc.zig b/src/hal/sdmmc.zig new file mode 100644 index 0000000..0bebeb0 --- /dev/null +++ b/src/hal/sdmmc.zig @@ -0,0 +1,2002 @@ +//! The SDMMC host controller, driven as an **SDIO host**. +//! +//! There is no SD card on this board. Slot 1 of the P4's SDMMC controller goes to an ESP32-C6 +//! running ESP-Hosted coprocessor firmware, which presents itself as a 4-bit SDIO device: CLK 18, +//! CMD 19, D0-D3 = 14/15/16/17. So this file implements CMD0/CMD5/CMD3/CMD7 and then CMD52/CMD53, +//! and nothing above them. SD memory cards, SPI mode, CSD/CID decoding and block devices are +//! deliberately absent - they are a different problem that happens to share a peripheral. +//! +//! The controller is a Synopsys DesignWare mobile-storage host. Three of its properties decide the +//! shape of everything below. +//! +//! **The card clock is not the register clock.** CLKDIV, CLKSRC and CLKENA are written on the bus +//! side and do not reach the card-interface unit until a *clock update command* is issued: a write +//! to the CMD register with `update_clk_reg` and `start_command` set, which sends nothing to the +//! card (`sdmmc_reg.h:440-456`, and ESP-IDF's `sd_host_slot_clock_update_command`, +//! `sd_host_sdmmc.c:896-912`). A driver that programmes a divider and moves on has changed +//! nothing. Three such commands are needed to change frequency safely - clock off, reprogramme, +//! clock on - and that is what `setBusClock` does. +//! +//! **The command register is a single word, and `start_command` is bit 31 of it.** Every attribute +//! of a command - index, whether a response is expected, whether its CRC is checked, whether data +//! follows and in which direction - is a field of the same word, and writing that word with bit 31 +//! set launches the command. So the interesting part of "send CMD52" is an encoding, not a +//! sequence, and `commandWord` is a pure function of the request. It is host-tested, and the +//! oracle compares the words it produces against words built through ESP-IDF's own +//! `sdmmc_hw_cmd_t` bitfields. +//! +//! **Data moves by internal DMA over descriptors in memory, and the P4 caches that memory.** +//! `soc_caps.h:185` sets SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE, so L2MEM - where every static in this +//! image lives - is reached by the CPU through the L1 data cache while the IDMAC reaches it +//! directly. See the "Cache" section below for the resolution; it is the one place in this file +//! where the right answer is not visible in any register header. +//! +//! Nothing here has been run on hardware by the author of this file. What is claimed is that the +//! register arithmetic and the command encodings match ESP-IDF's at the cited lines, that +//! `src/oracle/sdmmc_cases.zig` compares the two on the die, and that the configuration `init` +//! leaves behind reproduces a dump taken from a working ESP-IDF image on this board. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); +const intr = @import("intr.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +pub const Error = error{ Timeout, CrcError, ResponseError, NotSupported, Busy }; + +// ------------------------------------------------------------------------------- registers +// +// One instance, at DR_REG_SDHOST_BASE = DR_REG_SDMMC_BASE = 0x50083000 (`reg_base.h:44`, `:204`, +// and `esp32p4.peripherals.ld:41` agrees). The macros are spelled SDHOST_*, the peripheral is +// spelled SDMMC, and both names are ESP-IDF's. + +const ctrl = Reg.at(regs.SDHOST_CTRL_REG); +const clkdiv = Reg.at(regs.SDHOST_CLKDIV_REG); +const clksrc = Reg.at(regs.SDHOST_CLKSRC_REG); +const clkena = Reg.at(regs.SDHOST_CLKENA_REG); +const tmout = Reg.at(regs.SDHOST_TMOUT_REG); +const ctype = Reg.at(regs.SDHOST_CTYPE_REG); +const blksiz = Reg.at(regs.SDHOST_BLKSIZ_REG); +const bytcnt = Reg.at(regs.SDHOST_BYTCNT_REG); +const intmask = Reg.at(regs.SDHOST_INTMASK_REG); +const cmdarg = Reg.at(regs.SDHOST_CMDARG_REG); +const cmd = Reg.at(regs.SDHOST_CMD_REG); +const resp0 = Reg.at(regs.SDHOST_RESP0_REG); +const rintsts = Reg.at(regs.SDHOST_RINTSTS_REG); +/// The *masked* status: RINTSTS gated by INTMASK, and the only word the controller's interrupt +/// output looks at. ESP-IDF's `sdmmc_ll_get_intr_status` reads this one and not RINTSTS +/// (`sdmmc_ll.h:841-844`), which is exactly why INTMASK decides what reaches the CLIC while +/// RINTSTS stays readable for the polling path. +const mintsts = Reg.at(regs.SDHOST_MINTSTS_REG); +const status = Reg.at(regs.SDHOST_STATUS_REG); +const fifoth = Reg.at(regs.SDHOST_FIFOTH_REG); +const bmod = Reg.at(regs.SDHOST_BMOD_REG); +const pldmnd = Reg.at(regs.SDHOST_PLDMND_REG); +const dbaddr = Reg.at(regs.SDHOST_DBADDR_REG); +const idsts = Reg.at(regs.SDHOST_IDSTS_REG); +const idinten = Reg.at(regs.SDHOST_IDINTEN_REG); + +// CTRL fields. Two of them - `dma_enable` at bit 5 and `use_internal_dma` at bit 25 - have no +// `_S`/`_V` macro pair in `sdmmc_reg.h` at all: that header documents CTRL as bits 0,1,2,4,6..11 +// and stops. They are real, they are in the measured working dump (`ctrl=0x02000030`), and +// `sdmmc_struct.h:76` and `:135` name them at exactly those positions. This is the same situation +// as the IO MUX pull bits in `hal/gpio.zig`, and the same remedy: `Field.bit` with the struct +// header cited, because the struct header is ESP-IDF's definition of the layout even where the +// macro header is incomplete. +const controller_reset = Field.of(regs.SDHOST_CONTROLLER_RESET_S, regs.SDHOST_CONTROLLER_RESET_V); +const fifo_reset = Field.of(regs.SDHOST_FIFO_RESET_S, regs.SDHOST_FIFO_RESET_V); +const dma_reset = Field.of(regs.SDHOST_DMA_RESET_S, regs.SDHOST_DMA_RESET_V); +const int_enable = Field.of(regs.SDHOST_INT_ENABLE_S, regs.SDHOST_INT_ENABLE_V); +/// `sdmmc_struct.h:76` - `uint32_t dma_enable:1;` immediately after `int_enable:1` at bit 4. +const dma_enable = Field.bit(5); +/// `sdmmc_struct.h:135` - after `reserved2:4`, `card_voltage_a:4`, `card_voltage_b:4` and +/// `enable_od_pullup:1`, i.e. bit 25. `sdmmc_ll_enable_dma` (`sdmmc_ll.h:812-818`) is the only +/// writer, and the working dump's `ctrl=0x02000030` has exactly this bit plus 4 and 5. +const use_internal_dma = Field.bit(25); + +const clk_divider0 = Field.of(regs.SDHOST_CLK_DIVIDER0_S, regs.SDHOST_CLK_DIVIDER0_V); +const clk_divider1 = Field.of(regs.SDHOST_CLK_DIVIDER1_S, regs.SDHOST_CLK_DIVIDER1_V); +// CLKSRC is documented as one 4-bit field, two bits per card ("bit[1:0] are assigned for card 0, +// bit[3:2] are assigned for card 1", `sdmmc_reg.h:166-179`). `sdmmc_struct.h:191-192` splits it +// into `card0:2` and `card1:2`, which is the shape a driver wants; there are no macros for the +// halves, so the two sub-fields are spelled out with that citation. +const clksrc_card0 = Field.of(0, 0x3); +const clksrc_card1 = Field.of(2, 0x3); +const cclk_enable = Field.of(regs.SDHOST_CCLK_ENABLE_S, regs.SDHOST_CCLK_ENABLE_V); +const lp_enable = Field.of(regs.SDHOST_LP_ENABLE_S, regs.SDHOST_LP_ENABLE_V); +const response_timeout = Field.of(regs.SDHOST_RESPONSE_TIMEOUT_S, regs.SDHOST_RESPONSE_TIMEOUT_V); +const data_timeout = Field.of(regs.SDHOST_DATA_TIMEOUT_S, regs.SDHOST_DATA_TIMEOUT_V); +const card_width4 = Field.of(regs.SDHOST_CARD_WIDTH4_S, regs.SDHOST_CARD_WIDTH4_V); +const card_width8 = Field.of(regs.SDHOST_CARD_WIDTH8_S, regs.SDHOST_CARD_WIDTH8_V); +const block_size = Field.of(regs.SDHOST_BLOCK_SIZE_S, regs.SDHOST_BLOCK_SIZE_V); +const byte_count = Field.of(regs.SDHOST_BYTE_COUNT_S, regs.SDHOST_BYTE_COUNT_V); +const int_mask = Field.of(regs.SDHOST_INT_MASK_S, regs.SDHOST_INT_MASK_V); +const sdio_int_mask = Field.of(regs.SDHOST_SDIO_INT_MASK_S, regs.SDHOST_SDIO_INT_MASK_V); +const data_busy = Field.of(regs.SDHOST_DATA_BUSY_S, regs.SDHOST_DATA_BUSY_V); +const tx_wmark = Field.of(regs.SDHOST_TX_WMARK_S, regs.SDHOST_TX_WMARK_V); +const rx_wmark = Field.of(regs.SDHOST_RX_WMARK_S, regs.SDHOST_RX_WMARK_V); +const dma_msize = Field.of(regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_S, regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_V); +const bmod_swr = Field.of(regs.SDHOST_BMOD_SWR_S, regs.SDHOST_BMOD_SWR_V); +const bmod_fb = Field.of(regs.SDHOST_BMOD_FB_S, regs.SDHOST_BMOD_FB_V); +const bmod_de = Field.of(regs.SDHOST_BMOD_DE_S, regs.SDHOST_BMOD_DE_V); +const idinten_ti = Field.of(regs.SDHOST_IDINTEN_TI_S, regs.SDHOST_IDINTEN_TI_V); +const idinten_ri = Field.of(regs.SDHOST_IDINTEN_RI_S, regs.SDHOST_IDINTEN_RI_V); +const idinten_ni = Field.of(regs.SDHOST_IDINTEN_NI_S, regs.SDHOST_IDINTEN_NI_V); + +// The host-side clock generator, which is *not* in the SDMMC block: the P4 moved it into +// HP_SYS_CLKRST, and it is the first of two divider stages (this one, then CLKDIV inside the +// controller). `sdmmc_ll.h:227-228` for the source mux and gate, `:244-258` for the divider, +// `:305-315` for the sampling/driving phase clocks. +const peri_clk_ctrl01 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL01_REG); +const peri_clk_ctrl02 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL02_REG); + +const sdio_hs_mode = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_S, regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_V); +const sdio_ls_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_V); +const sdio_ls_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_V); +const sdio_ls_clk_edge_cfg_update = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_V); +const sdio_ls_clk_edge_l = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_V); +const sdio_ls_clk_edge_h = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_V); +const sdio_ls_clk_edge_n = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_V); +const sdio_ls_slf_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_V); +const sdio_ls_drv_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_V); +const sdio_ls_sam_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_V); +const sdio_ls_slf_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_V); +const sdio_ls_drv_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_V); +const sdio_ls_sam_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_V); + +// --------------------------------------------------------------------------------- interrupts +// +// RINTSTS / INTMASK share one 16-bit layout plus a 2-bit per-card SDIO field at [17:16]. +// `sdmmc_ll.h:35-53` names every bit; the numbers below are those, not a re-derivation. + +pub const Event = struct { + pub const cd: u32 = 1 << 0; // card detect + pub const re: u32 = 1 << 1; // response error + pub const cmd_done: u32 = 1 << 2; + pub const dto: u32 = 1 << 3; // data transfer over + pub const txdr: u32 = 1 << 4; + pub const rxdr: u32 = 1 << 5; + pub const rcrc: u32 = 1 << 6; // response CRC error + pub const dcrc: u32 = 1 << 7; // data CRC error + pub const rto: u32 = 1 << 8; // response timeout + pub const drto: u32 = 1 << 9; // data read timeout + pub const hto: u32 = 1 << 10; // data starvation by host timeout + pub const frun: u32 = 1 << 11; // FIFO under/overrun + pub const hle: u32 = 1 << 12; // hardware locked write error + pub const sbe: u32 = 1 << 13; // RX start-bit error + pub const acd: u32 = 1 << 14; // auto command done + pub const ebe: u32 = 1 << 15; // end-bit error + pub const io_slot0: u32 = 1 << 16; + pub const io_slot1: u32 = 1 << 17; + + /// What `sdmmc_ll.h:64-69` (SDMMC_LL_EVENT_DEFAULT) enables at init. Kept exactly as ESP-IDF + /// spells it, because the oracle compares against it; what this driver actually unmasks is + /// `armed`, below. + pub const default: u32 = cd | re | cmd_done | dto | rcrc | dcrc | rto | drto | hto | hle | sbe | ebe; + + /// `default` without card detect, and the only mask `configureInterrupts` ever writes. + /// + /// Bit 0 has to go, and this is not a preference. There is no card-detect pin on this board: + /// `configurePins` ties the signal to a matrix constant 0 ("card present"), and the + /// transition it makes while doing so *latches* RINTSTS.cd. RINTSTS is a sticky + /// write-1-to-clear register and nothing in the command path clears bit 0 - `sendCommand` + /// deliberately writes `default & ~cd` so as not to disturb asynchronous events. So with cd + /// unmasked, the controller's single output line into the CLIC is asserted from bring-up + /// onwards and never deasserts, and anyone who enables that CLIC line takes an interrupt + /// storm that no handler can end. Found by RxPath on CLIC line 21; the fix belongs here + /// rather than in the handler, because a level output that nothing can lower is this file's + /// bug. + /// + /// The two SDIO card-interrupt bits are absent from both masks: `setSlaveInterruptEnabled` + /// turns the one for this slot on when somebody is prepared to service it. + pub const armed: u32 = default & ~cd; + + /// Anything in here means the command failed. `sdmmc_ll.h:71-77` calls the superset + /// SDMMC_LL_SD_EVENT_MASK; this is the error half of it. + pub const command_errors: u32 = re | rcrc | rto | hle; + pub const data_errors: u32 = dcrc | drto | hto | frun | sbe | ebe; +}; + +/// The IDMAC's five reportable events - TI, RI, FBE, DU, CES - as one mask. `sdmmc_ll.h:83` +/// SDMMC_LL_EVENT_DMA_MASK. +const idsts_event_mask: u32 = 0x1f; + +/// The CLIC source this controller raises, for a caller that wants to be woken rather than to +/// poll. Registering a handler is `hal.intr`'s job and not this file's: see the note on +/// `slaveInterruptPending`. +pub const interrupt_source = intr.Source.sdio_host; + +// -------------------------------------------------------------------------------------- cache +// +// The IDMAC reads its descriptors and its data buffer straight out of L2MEM. The CPU reaches the +// same L2MEM through the L1 data cache (`soc_caps.h:185`, SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE), and +// that cache is write-back: `esp_cache_msync(..., DIR_C2M)` exists precisely because a store the +// CPU has made may still be sitting in a dirty line when the DMA engine reads memory. +// +// ESP-IDF offers two ways out and uses both. `sd_trans_sdmmc.c:135-139` writes descriptors through +// the normal address and calls `esp_cache_msync` after every one. `gdma_link.c:100-118` does it +// the other way: one write-back-and-invalidate when the region is created, and from then on every +// CPU access goes through the non-cacheable alias at `addr + 0x40000000` +// (`hal/cache_ll.h:27` CACHE_LL_L2MEM_NON_CACHE_ADDR, `soc/ext_mem_defs.h:68`). +// +// **This file takes the second route.** It is the cheaper one - no cache call in the transfer +// path - and it is the only one that stays correct without a cache HAL this project does not have. +// The one-time write-back-and-invalidate is still required, and skipping it is a real bug rather +// than a theoretical one: `_start` clears .bss with ordinary stores (`src/main.zig:85-92`), so +// every word of the DMA region below starts life as a *dirty* cache line full of zeros. Nothing +// says when those lines are evicted; if one is written back after a descriptor has been prepared +// through the alias, the descriptor becomes zero and the IDMAC stalls on an unowned descriptor. +// `gdma_link.c:107-112` does exactly this call for exactly this reason. +// +// The two ROM entry points are addressed directly rather than declared `extern`, because the +// generated linker script provides only `ets_printf` and `ets_delay_us`. The addresses are +// ESP-IDF's, from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld:186` and `:190` - the hw_ver1 +// file, which is the one that matches this die. (If they move into the linker script beside the +// other two, these two lines become `extern fn` and nothing else changes.) + +/// `soc/ext_mem_defs.h:68` SOC_NON_CACHEABLE_OFFSET. +pub const non_cacheable_offset: u32 = 0x4000_0000; + +/// `cache_ll_l1_dcache_get_line_size` reports this on the P4, and `sdmmc_struct.h:36-38` states it +/// in prose: "On P4, L1 Cache alignment is 64B". +pub const cache_line: u32 = 64; + +/// `rom/cache.h:230` - CACHE_MAP_L1_DCACHE is BIT(4). +const cache_map_l1_dcache: u32 = 1 << 4; + +const romCacheWriteBackAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int = + @ptrFromInt(0x4fc0_03f4); +const romCacheInvalidateAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int = + @ptrFromInt(0x4fc0_03e4); + +// ------------------------------------------------------------------------------- DMA descriptor + +/// One IDMAC descriptor, exactly as the hardware reads it: `sdmmc_struct.h:13-41`. +/// +/// ESP-IDF's `sdmmc_desc_t` is 64 bytes, not 16, and its own comment says why and when not to: +/// "These `reserved[12]` are for cache alignment... For those who want to access the DMA +/// descriptor in a non-cacheable way, you can consider remove these `reserved[12]` bytes" +/// (`sdmmc_struct.h:35-39`). That is this file, so the padding is gone and the descriptor is the +/// 16 bytes the IDMAC actually fetches. +pub const Descriptor = extern struct { + flags: u32, + /// [12:0] buffer1_size, [25:13] buffer2_size. + sizes: u32, + buffer1: u32, + /// Also `buffer2_ptr`; which one it is depends on `second_address_chained`. + next: u32, + + pub const disable_int_on_completion: u32 = 1 << 1; + pub const last_descriptor: u32 = 1 << 2; + pub const first_descriptor: u32 = 1 << 3; + pub const second_address_chained: u32 = 1 << 4; + pub const end_of_ring: u32 = 1 << 5; + pub const card_error_summary: u32 = 1 << 30; + pub const owned_by_idmac: u32 = 1 << 31; + + /// `sdmmc_struct.h:43` SDMMC_DMA_MAX_BUF_LEN. `buffer1_size` is 13 bits wide, so 8191 would + /// fit; ESP-IDF splits at 4096 and so does the bound below. + pub const max_buffer_len: u32 = 4096; +}; + +/// Bytes of L2MEM this driver owns, and the whole of its dynamic memory: there is no allocator +/// here and no allocation anywhere in the transfer path. +/// +/// 2 KiB of payload is chosen against what sits above: ESP-Hosted's SDIO transport moves at most +/// one 1600-byte frame plus its 12-byte header per CMD53, and the largest single command this +/// driver can express in block mode is 4 blocks of 512. Anything larger is split across commands +/// by `transferChunked`, which is correct for both addressing modes, so the number is a +/// speed/footprint trade and not a limit. +pub const bounce_len: u32 = 2048; + +/// Descriptor and bounce buffer in one cache-line-aligned region, so the one-time maintenance call +/// is one call over one range whose base and length are both multiples of 64. +const DmaRegion = extern struct { + desc: Descriptor, + _pad: [cache_line - @sizeOf(Descriptor)]u8, + buf: [bounce_len]u8, +}; + +comptime { + std.debug.assert(@sizeOf(Descriptor) == 16); + std.debug.assert(@sizeOf(DmaRegion) % cache_line == 0); + // One descriptor is enough only while the bounce buffer fits in one. If `bounce_len` ever + // grows past 4096 this has to become a ring, and this line is what will say so. + std.debug.assert(bounce_len <= Descriptor.max_buffer_len); +} + +/// 2112 bytes: 16 of descriptor, 48 of padding to a cache line, 2048 of payload. +var dma: DmaRegion align(cache_line) = std.mem.zeroes(DmaRegion); + +/// Addresses are `usize` rather than `u32` all the way to the register write. On this target the +/// two are the same type; on the host, where the arithmetic in these helpers is unit-tested, +/// `@intCast` of a real 64-bit address would panic before the test could check anything. +inline fn cachedAddr(p: *const anyopaque) usize { + return @intFromPtr(p); +} + +/// The address the *CPU* must use for anything in the DMA region. The hardware gets the cached +/// address - that is not an inconsistency, it is what ESP-IDF does: `gdma_link.c:268-273` hands +/// `list->items` to the peripheral and `:159` writes through `list->items_nc`. The alias exists to +/// change how the CPU's loads and stores are treated, and a bus master is not the CPU. +inline fn uncachedAddr(p: *const anyopaque) usize { + return cachedAddr(p) +% @as(usize, non_cacheable_offset); +} + +/// An address as the 32-bit register field the hardware reads it through. +inline fn busAddr(p: *const anyopaque) u32 { + return @intCast(cachedAddr(p)); +} + +inline fn descNc() *volatile Descriptor { + return @ptrFromInt(uncachedAddr(&dma.desc)); +} + +inline fn bufNc() [*]volatile u8 { + return @ptrFromInt(uncachedAddr(&dma.buf)); +} + +/// Write back and invalidate the DMA region once, so that no dirty line from `_start`'s .bss clear +/// can later land on top of what the alias writes. After this, the cached alias of this region is +/// never touched again by anything in this file. +fn syncDmaRegionOnce() void { + const base = busAddr(&dma); + const len: u32 = @sizeOf(DmaRegion); + _ = romCacheWriteBackAddr(cache_map_l1_dcache, base, len); + _ = romCacheInvalidateAddr(cache_map_l1_dcache, base, len); +} + +// -------------------------------------------------------------------------------------- timing +// +// Every wait in this file is bounded, and bounded in time rather than in loop iterations: a spin +// count is a different number on every optimize level, and this board has no debugger, so a wait +// that never returns is indistinguishable from a crash. +// +// The timebase is the RISC-V `cycle` CSR, the unprivileged shadow of `mcycle`, which is what +// ESP-IDF itself reads on this part (`rv_utils.h`, because SOC_CPU_HAS_CSR_PC is not defined for +// the P4) and what `src/soc.zig:116-131` already uses. It is deliberately *not* `hal.systimer`: +// systimer's `init` pulses the peripheral's reset, which would make the timebase jump under any +// other user, and `systimer.read` returns null when nothing has brought it up - neither is a +// property a bus driver should impose on its caller. +// +// The CPU clock is whatever the bootloader left, measured at 90 MHz on this board and rated to +// 400. Deadlines are computed at the 400 MHz *ceiling*, so on real silicon every timeout below is +// between 1x and 4.4x longer than its nominal microseconds. That is the safe direction: a timeout +// that fires early would turn a slow card into a spurious failure, and a timeout 4x long still +// terminates. +const assumed_cpu_hz_max: u32 = 400_000_000; + +inline fn cycleLow() u32 { + return asm volatile ("csrr %[r], 0xC00" + : [r] "=r" (-> u32), + ); +} + +/// A bounded wait. 32 bits of cycle counter wrap after 10.7 s at the assumed ceiling, which is an +/// order of magnitude past the longest deadline here, and the wrapping subtraction is correct +/// across the wrap anyway. +const Deadline = struct { + start: u32, + budget: u32, + + inline fn init(us: u32) Deadline { + return .{ .start = cycleLow(), .budget = us *% (assumed_cpu_hz_max / 1_000_000) }; + } + + inline fn expired(self: Deadline) bool { + return (cycleLow() -% self.start) >= self.budget; + } +}; + +/// `sd_host_private.h:62` SD_HOST_SDMMC_RESET_TIMEOUT_US. +const reset_timeout_us: u32 = 5_000_000; +/// `sd_host_private.h:61` SD_HOST_SDMMC_START_CMD_TIMEOUT_US - how long the CIU may take to accept +/// a command word, which is a bus-side handshake and nothing to do with the card. +const start_cmd_timeout_us: u32 = 1_000_000; +/// How long to wait for the card's response after the command has been accepted. The controller +/// has its own response timeout (TMOUT.response_timeout, 255 card clocks) and raises RTO, so this +/// only has to cover the case where the controller itself never reports anything. +const command_done_timeout_us: u32 = 200_000; +/// Data phase. TMOUT.data_timeout is programmed to 100 ms of card clocks, matching +/// `sd_host_sdmmc.c:531-533`; this outer bound is twice that. +const data_done_timeout_us: u32 = 200_000; +/// How long the card may hold DAT0 low before a new data command. +const busy_timeout_us: u32 = 500_000; + +// ------------------------------------------------------------------------------------- geometry + +pub const Width = enum { one, four }; + +/// Slot 1's pads on this board, and the GPIO-matrix signal each carries. +/// +/// Slot 0 has a direct IO MUX function and slot 1 does not +/// (`sdmmc_ll.h:88` SDMMC_LL_SLOT_SUPPORT_GPIO_MATRIX(1) is 1, and `sdmmc_periph.c:37-49` has +/// -1 for every slot-1 IO MUX pin), so every slot-1 signal is routed through the matrix. The +/// indices are `gpio_sig_map.h:8-18`, reached here through `regs` rather than written out: the +/// same discipline `hal/gpio.zig` applies to SIG_GPIO_OUT_IDX, for the same reason. +pub const Pins = struct { + clk: u8, + cmd: u8, + d0: u8, + d1: u8, + d2: u8, + d3: u8, +}; + +/// The ESP32-C6 coprocessor's wiring on this board. CLK 18, CMD 19, D0-D3 = 14/15/16/17. +pub const c6_pins: Pins = .{ .clk = 18, .cmd = 19, .d0 = 14, .d1 = 15, .d2 = 16, .d3 = 17 }; + +const sig = struct { + const cclk: u32 = @intCast(regs.SD_CARD_CCLK_2_PAD_OUT_IDX); + const ccmd: u32 = @intCast(regs.SD_CARD_CCMD_2_PAD_OUT_IDX); + const cdata0: u32 = @intCast(regs.SD_CARD_CDATA0_2_PAD_OUT_IDX); + const cdata1: u32 = @intCast(regs.SD_CARD_CDATA1_2_PAD_OUT_IDX); + const cdata2: u32 = @intCast(regs.SD_CARD_CDATA2_2_PAD_OUT_IDX); + const cdata3: u32 = @intCast(regs.SD_CARD_CDATA3_2_PAD_OUT_IDX); + const card_detect: u32 = @intCast(regs.SD_CARD_DETECT_N_2_PAD_IN_IDX); + const card_int: u32 = @intCast(regs.SD_CARD_INT_N_2_PAD_IN_IDX); + + comptime { + // The `_2` in these names is slot 1: `sdmmc_periph.c:52-76` fills + // `sdmmc_slot_gpio_sig[1]` from exactly these macros. Slot 0's set is named `_1` and would + // route the wrong controller port to the C6's pads, silently. + std.debug.assert(cclk == 0 and ccmd == 1 and cdata0 == 2); + std.debug.assert(cdata1 == 3 and cdata2 == 4 and cdata3 == 5); + + // The card interrupt is sensed on D1's *input* index, and `configurePins` hands `matrixIn` + // the *output* one - correct only because the P4's two signal tables agree on this signal. + // `gpio_sig_map.h:13-14` gives cdata1 the number 3 in both directions, and ESP-IDF relies + // on the same coincidence: `configure_pin_gpio_matrix` (`sd_host_sdmmc.c:1091-1105`) passes + // one `gpio_matrix_sig` to both `esp_rom_gpio_connect_in_signal` and `..._out_signal`. + // Asserted rather than assumed, because a mismatch here would route data correctly and + // sense interrupts from the wrong pad - which is invisible until something waits. + std.debug.assert(cdata1 == @as(u32, @intCast(regs.SD_CARD_CDATA1_2_PAD_IN_IDX))); + } +}; + +// -------------------------------------------------------------------------------------- state + +const State = struct { + slot: u1 = 1, + width: Width = .four, + /// The frequency `cardInit` switches to once the card is addressed and in 4-bit mode. + target_khz: u32 = 40_000, + pins: Pins = c6_pins, + /// Relative card address from CMD3, needed as the argument of CMD7. + rca: u16 = 0, + initialised: bool = false, +}; + +var state: State = .{}; + +/// The card's relative address, as returned by CMD3. Zero until `cardInit` has run. +pub fn rca() u16 { + return state.rca; +} + +inline fn slotBit() u32 { + return @as(u32, 1) << state.slot; +} + +// ------------------------------------------------------------------------------ command words +// +// One word, one function, no hardware. This is the part of the driver most worth testing on the +// host, and the part the oracle can compare against ESP-IDF's own bitfield struct without going +// anywhere near the card. + +/// Compose one field's contribution to a register word. `mmio.Reg.write` does this against a +/// register; here the destination is a value, because the command word is built, checked and only +/// then stored. +inline fn bits(comptime f: Field, v: u32) u32 { + return (v & f.unshiftedMask()) << f.shift; +} + +const cmd_index = Field.of(regs.SDHOST_CMD_INDEX_S, regs.SDHOST_CMD_INDEX_V); +const response_expect = Field.of(regs.SDHOST_RESPONSE_EXPECT_S, regs.SDHOST_RESPONSE_EXPECT_V); +const response_length = Field.of(regs.SDHOST_RESPONSE_LENGTH_S, regs.SDHOST_RESPONSE_LENGTH_V); +const check_response_crc = Field.of(regs.SDHOST_CHECK_RESPONSE_CRC_S, regs.SDHOST_CHECK_RESPONSE_CRC_V); +const data_expected = Field.of(regs.SDHOST_DATA_EXPECTED_S, regs.SDHOST_DATA_EXPECTED_V); +const read_write = Field.of(regs.SDHOST_READ_WRITE_S, regs.SDHOST_READ_WRITE_V); +const transfer_mode = Field.of(regs.SDHOST_TRANSFER_MODE_S, regs.SDHOST_TRANSFER_MODE_V); +const send_auto_stop = Field.of(regs.SDHOST_SEND_AUTO_STOP_S, regs.SDHOST_SEND_AUTO_STOP_V); +const wait_prvdata_complete = Field.of(regs.SDHOST_WAIT_PRVDATA_COMPLETE_S, regs.SDHOST_WAIT_PRVDATA_COMPLETE_V); +const stop_abort_cmd = Field.of(regs.SDHOST_STOP_ABORT_CMD_S, regs.SDHOST_STOP_ABORT_CMD_V); +const send_initialization = Field.of(regs.SDHOST_SEND_INITIALIZATION_S, regs.SDHOST_SEND_INITIALIZATION_V); +const card_number = Field.of(regs.SDHOST_CARD_NUMBER_S, regs.SDHOST_CARD_NUMBER_V); +const update_clock_registers_only = Field.of(regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_S, regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_V); +/// `sdmmc_reg.h:486-494` spells this `USE_HOLE_REG`; `sdmmc_struct.h:473` spells it +/// `use_hold_reg`, which is what it is - the hold register that synchronises CMD and DATA to +/// cclk_out. Same bit 29, and ESP-IDF sets it on every command (`sd_host_sdmmc.c:859-860`). +const use_hold_reg = Field.of(regs.SDHOST_USE_HOLE_REG_S, regs.SDHOST_USE_HOLE_REG_V); +const start_cmd = Field.of(regs.SDHOST_START_CMD_S, regs.SDHOST_START_CMD_V); + +pub const Response = enum { none, short, long }; +pub const Direction = enum { read, write }; + +/// Everything that distinguishes one command from another, in the terms the register uses. +pub const Command = struct { + index: u6, + response: Response = .none, + /// Whether the controller checks the response's CRC7. Off for R3 and R4, which do not carry a + /// valid one - `sd_protocol_types.h:140-141` define both without SCF_RSP_CRC, and + /// `make_hw_cmd` (`sd_trans_sdmmc.c:214-216`) keys `check_response_crc` off exactly that flag. + check_crc: bool = false, + data: ?Direction = null, + /// 80 clocks of 1 before the command. Required once after power-on, and set only for CMD0, + /// which is where ESP-IDF sets it (`sd_trans_sdmmc.c:197-206`). + send_init: bool = false, + /// Wait for a previous data transfer to finish before sending. Set on everything except CMD0, + /// CMD12 and CMD11, again following `make_hw_cmd`. + wait_prvdata: bool = true, + auto_stop: bool = false, + stop_abort: bool = false, + /// Not a command at all: push CLKDIV/CLKSRC/CLKENA into the card clock domain. + update_clock: bool = false, + slot: u1 = 0, +}; + +/// The 32-bit word that, written to SDHOST_CMD_REG, issues `c`. +/// +/// This is `make_hw_cmd` (`sd_trans_sdmmc.c:190-229`) plus the three fields +/// `sd_host_slot_start_command` adds afterwards - `use_hold_reg`, `card_num` and `start_command` +/// (`sd_host_sdmmc.c:859-881`) - because those three are not optional and splitting them across +/// two functions is how one of them gets forgotten. +pub fn commandWord(c: Command) u32 { + var w: u32 = 0; + w |= bits(cmd_index, c.index); + if (c.response != .none) w |= bits(response_expect, 1); + if (c.response == .long) w |= bits(response_length, 1); + if (c.check_crc) w |= bits(check_response_crc, 1); + if (c.data) |dir| { + w |= bits(data_expected, 1); + if (dir == .write) w |= bits(read_write, 1); + } + if (c.auto_stop) w |= bits(send_auto_stop, 1); + if (c.wait_prvdata) w |= bits(wait_prvdata_complete, 1); + if (c.stop_abort) w |= bits(stop_abort_cmd, 1); + if (c.send_init) w |= bits(send_initialization, 1); + if (c.update_clock) w |= bits(update_clock_registers_only, 1); + w |= bits(card_number, c.slot); + // Block transfers only; `transfer_mode` selects stream mode, which no SDIO command uses. + w |= bits(transfer_mode, 0); + w |= bits(use_hold_reg, 1); + w |= bits(start_cmd, 1); + return w; +} + +// ------------------------------------------------------------------------------ SDIO protocol +// +// Command indices and argument layouts, from `sd_protocol_defs.h`. Written out as constants rather +// than reached through `regs` because they are the SD specification, not this chip: the register +// headers know nothing about them. + +/// `sd_protocol_defs.h:35`, `:40`, `:61`, `:78-80`. +const cmd_go_idle_state: u6 = 0; +const cmd_send_relative_addr: u6 = 3; +const cmd_io_send_op_cond: u6 = 5; +const cmd_select_card: u6 = 7; +const cmd_io_rw_direct: u6 = 52; +const cmd_io_rw_extended: u6 = 53; + +/// CMD52's argument: `sd_protocol_defs.h:484-492`. +pub fn cmd52Arg(write: bool, func: u3, addr: u17, raw_flag: bool, data: u8) u32 { + var a: u32 = 0; + if (write) a |= @as(u32, 1) << 31; + a |= @as(u32, func) << 28; + if (raw_flag) a |= @as(u32, 1) << 27; + a |= @as(u32, addr) << 9; + a |= data; + return a; +} + +/// CMD53's argument: `sd_protocol_defs.h:496-506`. +/// +/// `count` is blocks in block mode and bytes in byte mode, and it is 9 bits: 0 means 512 in byte +/// mode ("See 5.3.1 SDIO simplified spec", `sdmmc_io.c:351-355`) and infinite in block mode, which +/// this driver never asks for. +pub fn cmd53Arg(write: bool, func: u3, addr: u17, block_mode: bool, incrementing: bool, count: u9) u32 { + var a: u32 = 0; + if (write) a |= @as(u32, 1) << 31; + a |= @as(u32, func) << 28; + if (block_mode) a |= @as(u32, 1) << 27; + if (incrementing) a |= @as(u32, 1) << 26; + a |= @as(u32, addr) << 9; + a |= count; + return a; +} + +/// The block size this driver programmes into BLKSIZ and into the card's CCCR/FBR. +/// `sdmmc_common.h:195` SDMMC_IO_BLOCK_SIZE, and ESP-Hosted writes the same 512 into FN0 and FN1 +/// (`port_esp_hosted_host_sdio.c:211-217`). +pub const io_block_size: u32 = 512; + +/// CCCR register offsets, `sd_protocol_defs.h:509-530`. +pub const cccr = struct { + pub const revision: u17 = 0x00; + pub const fn_enable: u17 = 0x02; + pub const fn_ready: u17 = 0x03; + pub const int_enable: u17 = 0x04; + pub const int_pending: u17 = 0x05; + pub const ctl: u17 = 0x06; + pub const bus_width: u17 = 0x07; + pub const card_cap: u17 = 0x08; + pub const cis_ptr: u17 = 0x09; + pub const blksize_l: u17 = 0x10; + pub const blksize_h: u17 = 0x11; + + pub const ctl_reset: u8 = 1 << 3; + pub const bus_width_1: u8 = 0; + pub const bus_width_4: u8 = 2; + /// Low-speed card; and "4-bit low speed", which says a low-speed card supports 4 bits anyway. + pub const card_cap_lsc: u8 = 1 << 6; + pub const card_cap_4bls: u8 = 1 << 7; +}; + +/// `sd_protocol_defs.h:533` SD_IO_FBR_START - function n's register block starts here. +const fbr_start: u17 = 0x100; + +/// R4's fields, `sd_protocol_defs.h:478-481`. +const r4_mem_ready: u32 = 1 << 31; +const r4_mem_present: u32 = 1 << 27; + +/// The voltage window the host offers in CMD5's second pass: bits 23:15, i.e. 2.8-3.6 V. +/// `sd_protocol_defs.h:109` SD_OCR_VOL_MASK, which is the whole of what `get_host_ocr` returns - +/// "For now tell that the host has 2.8-3.6V voltage range" (`sdmmc_common.h:174-180`). +const host_ocr: u32 = 0x00ff_8000; + +// ------------------------------------------------------------------------------- command issue + +/// Write one command word and wait for the CIU to take it. No card traffic is implied: a clock +/// update command goes through here too. +/// +/// Both waits are the ones `sd_host_slot_start_command` performs (`sd_host_sdmmc.c:862-892`), +/// bounded the same way. The first is not redundant with the second: writing any command register +/// while `start_command` is still set is a hardware locked write error, and HLE is reported +/// asynchronously in RINTSTS where it is easy to attribute to the wrong command. +fn startCommand(word: u32, arg: u32) Error!void { + var d = Deadline.init(start_cmd_timeout_us); + while (cmd.get(start_cmd) != 0) { + if (d.expired()) return error.Busy; + } + cmdarg.writeRaw(arg); + cmd.writeRaw(word); + d = Deadline.init(start_cmd_timeout_us); + while (cmd.get(start_cmd) != 0) { + if (d.expired()) return error.Timeout; + } +} + +/// Push CLKDIV, CLKSRC and CLKENA into the card clock domain. +fn clockUpdate() Error!void { + try startCommand(commandWord(.{ + .index = 0, + .update_clock = true, + .wait_prvdata = true, + .slot = state.slot, + }), 0); +} + +/// Turn a RINTSTS snapshot into the failure it describes. +/// +/// Order matters only in that the first match wins, and it is chosen so the most specific cause is +/// reported: a CRC error and a timeout together is a CRC error, because the timeout is downstream +/// of it. +fn decodeErrors(sts: u32) Error!void { + if (sts & (Event.rcrc | Event.dcrc) != 0) return error.CrcError; + if (sts & (Event.rto | Event.drto | Event.hto) != 0) return error.Timeout; + if (sts & (Event.re | Event.hle | Event.ebe | Event.sbe | Event.frun) != 0) return error.ResponseError; +} + +/// Wait for one or more RINTSTS bits, failing on any error bit or on the deadline. +/// +/// RINTSTS is write-1-to-clear, so this reads with `raw()` and clears with `writeRaw(mask)` - +/// never `modify`, which would clear every bit it read back and lose the events this function is +/// not waiting for. +fn waitEvents(want: u32, errors: u32, us: u32) Error!u32 { + const d = Deadline.init(us); + while (true) { + const sts = rintsts.raw(); + if (sts & errors != 0) { + rintsts.writeRaw(sts & (want | errors)); + try decodeErrors(sts & errors); + // Every bit any caller passes in `errors` is covered above; a new one arriving here + // is a bug in this file, and reporting it beats an `unreachable` on a board with no + // debugger. + return error.ResponseError; + } + if (sts & want == want) { + rintsts.writeRaw(want); + return sts; + } + if (d.expired()) return error.Timeout; + } +} + +/// A command with no data phase: issue it, wait for command-done, return R1/R5's first word. +fn sendCommand(c: Command, arg: u32) Error!u32 { + // Everything this command is about to overwrite. This slot's SDIO card interrupt is + // deliberately left alone - the C6 raises it asynchronously and clearing it here would drop a + // wakeup the layer above is waiting for - and `clearNonSlaveInterrupts` is exactly that set. + // + // It used to be `Event.default & ~Event.cd`, which is a *subset* of the event bits and left + // four of them latched for ever: txdr(4), rxdr(5), frun(11) and acd(14). Two consequences, one + // cosmetic and one not. Cosmetic: every RINTSTS a diagnostic prints carries a stale 0x10 from + // the first transfer onwards, which is noise in exactly the register that has to be read + // carefully. Not cosmetic: **frun is a member of `Event.data_errors`**, so one FIFO + // under/overrun - ever - would latch a bit that nothing clears and fail every subsequent + // `waitEvents(Event.dto, Event.data_errors, ...)` for the rest of the run. The data path works + // today only because frun has never fired. + clearNonSlaveInterrupts(); + var cc = c; + cc.slot = state.slot; + try startCommand(commandWord(cc), arg); + _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us); + return resp0.raw(); +} + +/// R5's status byte, the one CMD52 and CMD53 return. `sd_protocol_defs.h:493` takes the data byte; +/// the flags above it say whether the card accepted the command at all. +const r5_com_crc_error: u32 = 1 << 15; +const r5_illegal_command: u32 = 1 << 14; +const r5_error: u32 = 1 << 11; +const r5_function_number: u32 = 1 << 9; +const r5_out_of_range: u32 = 1 << 8; +const r5_bad: u32 = r5_com_crc_error | r5_illegal_command | r5_error | r5_function_number | r5_out_of_range; + +fn checkR5(r: u32) Error!u8 { + if (r & r5_com_crc_error != 0) return error.CrcError; + if (r & r5_bad != 0) return error.ResponseError; + return @truncate(r); +} + +// ------------------------------------------------------------------------------- bring-up + +/// Controller, FIFO and DMA reset, then wait for all three to self-clear. +/// +/// All three bits are self-clearing, and `sdmmc_ll.h:486`, `:510` and `:534` each say so with a +/// different delay ("two AHB clock cycles", "after reset done"). ESP-IDF sets all three and polls +/// all three together (`sd_host_sdmmc.c:917-950`), which is what makes one bounded wait correct +/// for the set. +pub fn resetController() Error!void { + ctrl.modify(.{ controller_reset.is(1), fifo_reset.is(1), dma_reset.is(1) }); + const d = Deadline.init(reset_timeout_us); + while (true) { + const v = ctrl.raw(); + if (v & (controller_reset.mask() | fifo_reset.mask() | dma_reset.mask()) == 0) return; + if (d.expired()) return error.Timeout; + } +} + +/// The interrupt configuration `sd_host_sdmmc.c:120-124` establishes - clear everything, mask +/// everything, then unmask the completion and error events and turn the global enable on - with +/// one deliberate deviation: card detect stays masked *and* gets cleared. See `Event.armed` for +/// why that bit is load-bearing on a board with no card-detect pin. +/// +/// `int_enable` gates the controller's single line into the CLIC. It is on even though this driver +/// polls, because RINTSTS is set regardless and the layer above may register a handler for the +/// SDIO card interrupt; leaving it off would mean `setSlaveInterruptEnabled(true)` silently did +/// nothing. +pub fn configureInterrupts() void { + rintsts.writeRaw(0xffff_ffff); + intmask.writeRaw(0); + ctrl.modify(.{int_enable.is(0)}); + intmask.writeRaw(Event.armed); + // Belt and braces: `armed` keeps the controller from reporting a latched cd, and this makes + // sure there is no latched cd to report if anything ever unmasks it again. + rintsts.writeRaw(Event.cd); + ctrl.modify(.{int_enable.is(1)}); +} + +/// `sdmmc_ll_init_dma`, `sdmmc_ll.h:796-804`: enable the DMA path, clear the bus-mode register, +/// pulse the IDMAC's own software reset, and unmask its three completion interrupts. +pub fn initDma() void { + ctrl.modify(.{dma_enable.is(1)}); + bmod.writeRaw(0); + bmod.modify(.{bmod_swr.is(1)}); + idinten.modify(.{ idinten_ni.is(1), idinten_ri.is(1), idinten_ti.is(1) }); +} + +/// Leave the controller's interrupt output silent, and both status registers clean. +/// +/// `configureInterrupts` and `initDma` above are ESP-IDF's sequences, and ESP-IDF is +/// interrupt-driven: its transfers wait on a queue its ISR fills, so it needs command-done, the +/// error bits and the IDMAC's completions in the masks. **This driver polls**, so every one of +/// those is noise on a line whose only handler understands one cause. Worse than noise: two of +/// them hold the line asserted forever. +/// +/// * **INTMASK** gates RINTSTS into MINTSTS. Zero here costs nothing - `waitEvents` reads +/// RINTSTS, and "Bits are logged regardless of interrupt mask status" +/// (`sdmmc_struct.h:589-591`). +/// * **IDINTEN** gates the IDMAC's own events, and it does *not* go through INTMASK. `initDma` +/// enables NI/RI/TI because `sdmmc_ll_init_dma` does, and IDF can afford that because its ISR +/// clears IDSTS on every interrupt (`sd_host_sdmmc.c:801-802`). `dataTransfer` clears IDSTS +/// *before* a transfer and nothing clears it after, so RI and its sticky summary NIS stay set +/// from the first CMD53 onwards - a permanently asserted interrupt line that no INTMASK write +/// can lower. +/// +/// `CTRL.int_enable` stays on: with both masks at zero the line cannot assert anyway, and leaving +/// the global enable alone keeps `armSlaveInterrupt` down to the stores that matter. +pub fn muteInterrupts() void { + intmask.writeRaw(0); + idinten.writeRaw(0); + rintsts.writeRaw(0xffff_ffff); + idsts.writeRaw(idsts_event_mask); +} + +/// FIFO watermarks and DMA burst size. +/// +/// ESP-IDF never writes this register on any target - there is no `sdmmc_ll` function for it and +/// no assignment anywhere in `components/` - so the value in the measured working dump, +/// `fifoth=0x01FF0000`, is the hardware's reset state: rx watermark 511, tx watermark 0, burst +/// size code 0 (one transfer). This function writes that value explicitly rather than inheriting +/// it, because a controller reset is not the only thing that can have touched the register and +/// "the same as reset" is a claim worth making in code. +/// +/// It is also a performance knob left deliberately untouched: DesignWare recommends half the FIFO +/// depth for both watermarks and a burst size matching the AXI port, and tx watermark 0 means a +/// DMA request only when the FIFO is completely empty. Turning that knob without a board to +/// measure on would be guessing, and the guess would be against a configuration known to work at +/// 40 MHz. +pub fn setFifoThreshold(rx: u32, tx: u32, msize: u32) void { + fifoth.write(.{ rx_wmark.is(rx), tx_wmark.is(tx), dma_msize.is(msize) }); +} + +/// The reset-value watermarks, which are the ones the working dump shows. +pub const default_rx_watermark: u32 = 511; +pub const default_tx_watermark: u32 = 0; +pub const default_dma_msize: u32 = 0; + +/// Bus width, host side. The card side is a CCCR write and is done in `cardInit`; the two must +/// change in that order, or the next command goes out on a bus the card is not listening to. +pub fn setBusWidth(w: Width) void { + const m = slotBit(); + const c8 = ctype.get(card_width8) & ~m; + const c4 = switch (w) { + .one => ctype.get(card_width4) & ~m, + .four => ctype.get(card_width4) | m, + }; + ctype.modify(.{ card_width4.is(c4), card_width8.is(c8) }); +} + +pub fn setBlockSize(bytes: u32) void { + blksiz.modify(.{block_size.is(bytes)}); +} + +/// The two-stage divider, resolved. Stage one is `host_div` in HP_SYS_CLKRST, stage two is the +/// controller's own CLKDIV, and the card clock is `160 MHz / host_div / (2 * card_div)` with +/// `card_div == 0` meaning bypass. +/// +/// The table is `sd_host_slot_get_clk_dividers` (`sd_host_sdmmc.c:998-1062`), restricted to the +/// PLL160M source: this board's C6 is a 3.3 V SDIO device, so the 200 MHz SDIO PLL and the UHS-I +/// speeds it exists for are out of reach and out of scope. +pub const Dividers = struct { host: u32, card: u32 }; + +pub fn dividersFor(khz: u32) Dividers { + const src_hz: u32 = 160_000_000; + if (khz >= 40_000) return .{ .host = 4, .card = 0 }; // 160/4 = 40 MHz + if (khz == 20_000) return .{ .host = 8, .card = 0 }; // 160/8 = 20 MHz + if (khz == 400) return .{ .host = 10, .card = 20 }; // 160/10/(20*2) = 400 kHz + var host = src_hz / (khz * 1000); + var card: u32 = 0; + if (host > 15) { + host = 2; + card = (src_hz / 2) / (2 * khz * 1000); + if (((src_hz / 2) % (2 * khz * 1000)) > 0) card += 1; + } else if (src_hz % (khz * 1000) > 0) { + host += 1; + } + return .{ .host = host, .card = card }; +} + +/// Stage one: the clock generator in HP_SYS_CLKRST. `sdmmc_ll_set_clock_div`, +/// `sdmmc_ll.h:244-258`. +/// +/// The `edge_cfg_update` bit is write-to-trigger and must be pulsed - set then cleared - after the +/// three edge fields, or the new division is programmed and never latched. +pub fn setHostClockDiv(div: u32) void { + if (div > 1) { + peri_clk_ctrl02.modify(.{ + sdio_ls_clk_edge_h.is(div / 2 - 1), + sdio_ls_clk_edge_n.is(div - 1), + sdio_ls_clk_edge_l.is(div - 1), + }); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)}); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)}); + } else { + peri_clk_ctrl01.modify(.{sdio_hs_mode.is(1)}); + peri_clk_ctrl02.modify(.{ + sdio_ls_clk_edge_h.is(0), + sdio_ls_clk_edge_n.is(0), + sdio_ls_clk_edge_l.is(0), + }); + } +} + +/// PLL160M, the only source this driver uses. `sdmmc_ll_select_clk_source`, `sdmmc_ll.h:212-229`: +/// source value 0 is PLL160M and 1 is the 200 MHz SDIO PLL. +pub fn selectPll160m() void { + peri_clk_ctrl01.modify(.{ sdio_ls_clk_src_sel.is(0), sdio_ls_clk_en.is(1) }); +} + +/// The driving, sampling and self clocks the pad logic runs on. `sdmmc_ll_init_phase_delay`, +/// `sdmmc_ll.h:303-315`. Without this the three gates stay off and the bus does not move, which is +/// the kind of failure that looks like a wiring fault. +pub fn initPhaseDelay() void { + peri_clk_ctrl02.modify(.{ + sdio_ls_drv_clk_en.is(1), + sdio_ls_sam_clk_en.is(1), + sdio_ls_slf_clk_en.is(1), + sdio_ls_drv_clk_edge_sel.is(1), + sdio_ls_sam_clk_edge_sel.is(0), + sdio_ls_slf_clk_edge_sel.is(0), + }); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)}); + peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)}); +} + +/// Stage one, whole: divider, source, phase clocks, and the settle the hardware needs afterwards. +/// `sd_host_set_clk_div`, `sd_host_sdmmc.c:974-990`, including its closing +/// `esp_rom_delay_us(10)` - "Wait for the clock to propagate". +/// +/// This has to happen before the controller reset, not after. `controller_reset` is documented to +/// self-clear "after two AHB and two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`), so with +/// no card clock reaching the block the bit never clears and the reset wait runs to its full +/// timeout. ESP-IDF's order says the same thing without saying it: `sd_host_set_clk_div` at +/// `sd_host_sdmmc.c:109`, `sd_host_reset` at `:112`. +pub fn setHostClock(div: u32) void { + setHostClockDiv(div); + selectPll160m(); + initPhaseDelay(); + spinMicros(10); +} + +/// Stage two: the controller's per-slot divider and the divider-to-slot mux. +/// `sdmmc_ll_set_card_clock_div`, `sdmmc_ll.h:431-442`. Slot 1 uses divider 1, slot 0 uses divider +/// 0 - so the mux value equals the slot number, which is why one line covers both. +pub fn setCardClockDiv(div: u32) void { + if (state.slot == 0) { + clksrc.modify(.{clksrc_card0.is(0)}); + clkdiv.modify(.{clk_divider0.is(div)}); + } else { + clksrc.modify(.{clksrc_card1.is(1)}); + clkdiv.modify(.{clk_divider1.is(div)}); + } +} + +/// The card clock's on/off switch, one bit per slot. Takes effect only after a clock update +/// command. `sdmmc_ll_enable_card_clock`, `sdmmc_ll.h:415-422`. +pub fn setCardClockEnabled(on: bool) void { + const cur = clkena.get(cclk_enable); + clkena.modify(.{cclk_enable.is(if (on) cur | slotBit() else cur & ~slotBit())}); +} + +/// Stop the card clock while the card is idle. `sdmmc_ll_enable_card_clock_low_power`, +/// `sdmmc_ll.h:474-481`. **Off** for SDIO: the card raises its interrupt on D1 and cannot do so +/// with the clock stopped, which is why ESP-IDF clears the same bit for any slot with +/// `cclk_always_on` (`sd_host_sdmmc.c:272-285`) and why the measured working dump reads +/// `clkena=0x00000002` rather than `0x00020002`. +pub fn setCardClockLowPower(on: bool) void { + const cur = clkena.get(lp_enable); + clkena.modify(.{lp_enable.is(if (on) cur | slotBit() else cur & ~slotBit())}); +} + +/// Bytes in the next data transfer. `sdmmc_ll_set_data_transfer_len`, `sdmmc_ll.h:651-654`. +pub fn setDataTransferLen(len: u32) void { + bytcnt.modify(.{byte_count.is(len)}); +} + +/// Data-read and response timeouts, both in card output clocks. +/// `sdmmc_ll_set_data_timeout` / `sdmmc_ll_set_response_timeout`, `sdmmc_ll.h:564-582`. +pub fn setTimeouts(data_cycles: u32, response_cycles: u32) void { + tmout.write(.{ + data_timeout.is(if (data_cycles > 0xff_ffff) 0xff_ffff else data_cycles), + response_timeout.is(response_cycles), + }); +} + +/// Turn the internal DMA path on or off: both CTRL bits and both BMOD bits, together. +/// `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`. +pub fn setDmaEnabled(on: bool) void { + const v: u32 = @intFromBool(on); + ctrl.modify(.{ dma_enable.is(v), use_internal_dma.is(v) }); + bmod.modify(.{ bmod_de.is(v), bmod_fb.is(v) }); +} + +/// Where the IDMAC fetches its first descriptor. `sdmmc_ll_set_desc_addr`, `sdmmc_ll.h:673-676`. +/// The address is the *cached* one; see the "Cache" section above for why that is right. +pub fn setDescriptorAddr(a: u32) void { + dbaddr.writeRaw(a); +} + +/// Change the card clock, safely: stop it, reprogramme both stages, start it again, with a clock +/// update command after each step. `sd_host_slot_set_card_clk`, `sd_host_sdmmc.c:487-537`. +/// +/// Low-power mode is left **off**, which is the one place this deviates from a plain SD host and +/// matches the measured dump (`clkena=0x00000002`: clock enabled for slot 1, `lp_enable` clear). +/// `clkena.lp_enable` stops cclk while the card is idle; an SDIO card signals its interrupt on D1 +/// and needs the clock running to do it, which is why ESP-IDF turns the same bit off for any slot +/// with `cclk_always_on` (`sd_host_sdmmc.c:272-285`). +pub fn setBusClock(khz: u32) Error!void { + const d = dividersFor(khz); + + setCardClockEnabled(false); + try clockUpdate(); + + setCardClockDiv(d.card); + setHostClock(d.host); + try clockUpdate(); + + setCardClockEnabled(true); + setCardClockLowPower(false); + try clockUpdate(); + + // 100 ms of card clocks for data, and the maximum 255 card clocks for a response - "always set + // response timeout to highest value, it's small enough anyway" (`sd_host_sdmmc.c:534-535`). + setTimeouts(100 * khz, 255); +} + +/// Route slot 1's six signals to the C6's pads. +/// +/// Pull-ups: **the board provides them externally and this enables the internal ones anyway**, on +/// all six pads, because that is what the working configuration does. It is not obvious from +/// ESP-Hosted's side - it leaves `SDMMC_SLOT_FLAG_INTERNAL_PULLUP` clear +/// (`SDMMC_SLOT_CONFIG_DEFAULT`, `sdmmc_default_configs.h:98`: `.flags = 0`) - but every pad still +/// gets one, because `configure_pin_gpio_matrix` opens with `gpio_reset_pin` +/// (`sd_host_sdmmc.c:1096`) and that function enables the pull-up unconditionally: "for powersave +/// reasons, the GPIO should not be floating, select pullup" (`gpio.c:469-472`). The 40 MHz link +/// that produced the register dump therefore had both the module's external pull-ups and these. +/// Matching a measured configuration beats reasoning about which resistor is redundant. +/// +/// D1 has a second job: it is the SDIO interrupt line, and the controller derives that interrupt +/// from the same routed data signal - `sd_host_slot_sdmmc_io_int_enable` (`sd_host_sdmmc.c:381-388`) +/// is *only* `configure_pin(d1, sdmmc_slot_gpio_sig[slot].d1, GPIO_MODE_INPUT_OUTPUT)`, the same +/// two matrix writes and the same `fun_ie` the loop below already does, and it touches no +/// controller register at all. Both halves are load-bearing and neither is visible in a working +/// data path: with `fun_ie` clear, or with the *input* side of the matrix left pointing elsewhere, +/// D1 still drives and every transfer still completes while the controller samples a constant and +/// never latches a card interrupt. A link that carries traffic and never reports an event is +/// exactly what that failure looks like, which is why D1 is routed both ways even in 1-bit mode +/// and why `interruptDiagnostics` prints both bits. +/// +/// D3 is *not* routed to the controller yet. It is driven high as a plain GPIO output until the +/// bus is switched to 4 bits, which is how a host tells an SDIO card to use SD mode rather than +/// SPI mode; `sd_host_sdmmc.c:1282-1294` does the same and `cardInit` reconnects it at +/// `sd_host_sdmmc.c:575-583`'s point in the sequence. +pub fn configurePins(pins: Pins) void { + // CLK is output-only. + gpio.matrixOut(pins.clk, sig.cclk); + gpio.setInputEnable(pins.clk, false); + gpio.setPull(pins.clk, .up); + + const bidir = [_]struct { pin: u8, signal: u32 }{ + .{ .pin = pins.cmd, .signal = sig.ccmd }, + .{ .pin = pins.d0, .signal = sig.cdata0 }, + .{ .pin = pins.d1, .signal = sig.cdata1 }, + .{ .pin = pins.d2, .signal = sig.cdata2 }, + }; + for (bidir) |b| { + gpio.matrixOut(b.pin, b.signal); + gpio.matrixIn(b.pin, b.signal); + gpio.setInputEnable(b.pin, true); + gpio.setPull(b.pin, .up); + } + + // D3 high, as a GPIO, until the bus width changes. + gpio.configureOutput(pins.d3, .{ .readback = true }); + gpio.setPull(pins.d3, .up); + gpio.setHigh(pins.d3); + + // Card detect and the card's own interrupt-request pin are not wired to anything on this + // board, so both are tied off in the matrix exactly as ESP-IDF ties them when no pin is + // configured: card-detect to a constant 0 ("card present", `sd_host_sdmmc.c:1315-1319`) and + // card-int-n to a constant 1, i.e. inactive (`:1304-1306`). Leaving them unrouted is not the + // same thing: GPIO_FUNCn_IN_SEL_CFG resets with `sig_in_sel` clear, which bypasses the matrix + // and takes the signal from whatever direct pad function exists - and slot 1 has none. + // Write protect is left alone; nothing in this driver reads WRTPRT. + gpio.matrixIn(gpio.matrix_const_zero, sig.card_detect); + gpio.matrixIn(gpio.matrix_const_one, sig.card_int); +} + +/// Reconnect D3 to the controller, once the card is in 4-bit mode. +fn attachD3() void { + gpio.matrixOut(state.pins.d3, sig.cdata3); + gpio.matrixIn(state.pins.d3, sig.cdata3); + gpio.setInputEnable(state.pins.d3, true); + gpio.setPull(state.pins.d3, .up); +} + +/// Everything from the clock gate to a controller that will accept a command, with the bus at the +/// 400 kHz probing frequency and 1 bit wide - which is where an SDIO card has to be met. +/// +/// `cardInit` is what raises it to `khz` and to `width`, after the card has been addressed. +pub fn init(opts: struct { + slot: u1 = 1, + width: Width = .four, + khz: u32 = 40_000, + pins: Pins = c6_pins, +}) Error!void { + state = .{ + .slot = opts.slot, + .width = opts.width, + .target_khz = opts.khz, + .pins = opts.pins, + }; + + // The C6 hangs off slot 1 and slot 0's pads are the P4's own flash on most boards; refusing + // here is cheaper than debugging a bricked boot. + if (opts.slot != 1) return error.NotSupported; + + // 1. Bus clock and reset. Unlike most of this chip, SDMMC's bus clock is gated *off* at + // power-on (HP_SYS_CLKRST SOC_CLK_CTRL1 REG_SDMMC_SYS_CLK_EN, default 0), so this is a + // prerequisite and not a formality - without it the register block reads stale nonsense. + // Its reset bit is not in HP_SYS_CLKRST at all but in LP_AON_CLKRST; see hal/clkrst.zig. + clkrst.init(.sdmmc); + + // 2. The host clock generator, *before* the controller reset and not after. `sd_host_reset` + // polls three self-clearing bits, and `controller_reset` clears only "after two AHB and + // two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`) - with no card clock reaching the + // block that poll runs to its full timeout. ESP-IDF's controller init has the same order: + // `sd_host_set_clk_div(ctlr, SDMMC_CLK_SRC_DEFAULT, 2)` at `sd_host_sdmmc.c:109`, then + // `sd_host_reset` at `:112`. Divider 2 is IDF's provisional value, replaced at step 6. + setHostClock(2); + + // 3. Controller, FIFO and DMA out of reset. + try resetController(); + + // 4. Interrupts and DMA, before any command can produce one. The first two reproduce ESP-IDF + // for the differential; `muteInterrupts` then takes back everything this driver polls for + // instead of being interrupted by, leaving the line into the CLIC silent until a waiter + // arms it. + configureInterrupts(); + initDma(); + muteInterrupts(); + + // 5. Pads. After the clock gate so the controller's outputs are real, before the card clock so + // the first cycle the C6 sees is a clean one. + configurePins(state.pins); + + // 6. Bus clock at probing speed, 1 bit wide. An SDIO card has to be met at 400 kHz in 1-bit + // mode; `cardInit` raises both once the card has been addressed. + try setBusClock(400); + setBusWidth(.one); + + // 7. Transfer geometry. + setBlockSize(io_block_size); + setFifoThreshold(default_rx_watermark, default_tx_watermark, default_dma_msize); + setDescriptorAddr(busAddr(&dma.desc)); + + // 8. The one cache operation in this driver's life. See the "Cache" section above. + syncDmaRegionOnce(); + + state.initialised = true; + + // The one claim in this sequence with no differential case behind it, said out loud once, at + // the moment it is true. `configureInterrupts` and `initDma` are compared against ESP-IDF on + // the die; `muteInterrupts` cannot be - it is a deliberate deviation from IDF's ISR-driven + // design, and a reference implementation of our own decision would prove nothing. This line is + // the substitute, and it is worth a print because both zeros are load-bearing: a non-zero + // idinten here is an interrupt line that no INTMASK write can ever lower. + note("MARK SDMMC_INIT intmask=0x%08x idinten=0x%08x expect 0x00000000 and 0x00000000\r\n", .{ + intmask.raw(), idinten.raw(), + }); +} + +// ------------------------------------------------------------------------------ card bring-up + +/// CMD0, CMD5, CMD3, CMD7, then the CCCR writes that make function 1 usable: the sequence that +/// takes the C6 from "powered" to "answers CMD52". +/// +/// The command half follows `sdmmc_card_init` (`sdmmc_init.c:78-133`) restricted to the SDIO path: +/// `sdmmc_io_reset`, CMD0, `sdmmc_init_io` (CMD5 twice), `sdmmc_init_rca` (CMD3), +/// `sdmmc_init_select_card` (CMD7). The CCCR half is ESP-Hosted's `hosted_sdio_card_fn_init` +/// (`port_esp_hosted_host_sdio.c:143-220`) - enable function 1, wait for it to report ready, +/// unmask its interrupt, switch to 4 bits, set both block sizes to 512 - because that is what this +/// particular device needs and IDF's generic SDIO init does not do. +/// +/// The CMD52 that resets the card is allowed to fail. A device that is already out of reset +/// answers it; one that is not may time out, and `sdmmc_io_reset` (`sdmmc_io.c:66-83`) accepts +/// exactly that. +pub fn cardInit() Error!void { + if (!state.initialised) return error.NotSupported; + + // CCCR CTL bit 3: I/O reset. Best-effort, as above. + cmd52Write(0, cccr.ctl, cccr.ctl_reset) catch {}; + + // CMD0 with the 80-clock init sequence and no response. + _ = try sendCommand(.{ + .index = cmd_go_idle_state, + .response = .none, + .send_init = true, + .wait_prvdata = false, + }, 0); + // SDMMC_GO_IDLE_DELAY_MS (`sdmmc_common.h:34`), which `sdmmc_send_cmd_go_idle_state` waits + // out before returning (`sdmmc_cmd.c:114-116`). CMD0 has no response, so there is nothing to + // wait *for*: this is the card's own settling time and skipping it makes the next command a + // coin toss. + spinMicros(20_000); + + // CMD5 with a zero argument asks "are you an IO card, and what voltages do you take"; R4 has + // no CRC, hence `check_crc = false` (`sd_protocol_types.h:141`). + const probe = try sendCommand(.{ + .index = cmd_io_send_op_cond, + .response = .short, + .check_crc = false, + }, 0); + const functions = (probe >> 28) & 0x7; + if (functions == 0) return error.NotSupported; // answered CMD5, but has no IO function + + // CMD5 again with the voltage window, until the card reports ready. 100 attempts is + // `sdmmc_io.c:240`; the 10 ms between them is SDMMC_IO_SEND_OP_COND_DELAY_MS + // (`sdmmc_common.h:35`), spent here as a bounded spin rather than a scheduler delay. + const ocr = host_ocr & probe; + var ready = false; + var tries: u32 = 0; + while (tries < 100) : (tries += 1) { + const r = try sendCommand(.{ + .index = cmd_io_send_op_cond, + .response = .short, + .check_crc = false, + }, ocr); + if (r & r4_mem_ready != 0) { + ready = true; + break; + } + spinMicros(10_000); + } + if (!ready) return error.Timeout; + + // CMD3: the card picks its own relative address and returns it in R6[31:16]. + const r6 = try sendCommand(.{ + .index = cmd_send_relative_addr, + .response = .short, + .check_crc = true, + }, 0); + state.rca = @truncate(r6 >> 16); + + // CMD7 with that address moves the card from stand-by to transfer state. Every CMD52 and + // CMD53 after this is addressed to it implicitly. + _ = try sendCommand(.{ + .index = cmd_select_card, + .response = .short, + .check_crc = true, + }, @as(u32, state.rca) << 16); + + // ---- CCCR: function 1 on. + const ioe = try cmd52Read(0, cccr.fn_enable); + try cmd52Write(0, cccr.fn_enable, ioe | 0x02); + + // Wait for IOR bit 1. ESP-Hosted polls with a 10 ms gap and gives up after SDIO_INIT_MAX_RETRY + // (`port_esp_hosted_host_sdio.c:177-192`). + var fn_ready = false; + tries = 0; + while (tries < 100) : (tries += 1) { + if ((try cmd52Read(0, cccr.fn_ready)) & 0x02 != 0) { + fn_ready = true; + break; + } + spinMicros(10_000); + } + if (!fn_ready) return error.Timeout; + + // Master interrupt enable plus function 1's, so the C6 can raise D1. + const ie = try cmd52Read(0, cccr.int_enable); + try cmd52Write(0, cccr.int_enable, ie | 0x01 | 0x02); + + // ---- Bus width: card first, then host, then D3 joins the bus. + if (state.width == .four) { + const cap = try cmd52Read(0, cccr.card_cap); + // "Not a low-speed card" or "a low-speed card that supports 4 bits" - `sdmmc_io.c:182-183`. + if ((cap & cccr.card_cap_lsc) == 0 or (cap & cccr.card_cap_4bls) != 0) { + try cmd52Write(0, cccr.bus_width, cccr.bus_width_4); + setBusWidth(.four); + attachD3(); + } else { + state.width = .one; + } + } + + // ---- Block size 512 for function 0 and function 1, host side and card side. + try setCardBlockSize(0, io_block_size); + try setCardBlockSize(1, io_block_size); + setBlockSize(io_block_size); + + // ---- Finally the target frequency, now that the card is addressed and the bus is wide. + try setBusClock(state.target_khz); +} + +/// The 16-bit block size lives in two consecutive byte registers, low half first +/// (`port_esp_hosted_host_sdio.c:123-141`). Function n's copy is at `0x100 * n + 0x10`. +fn setCardBlockSize(func: u3, bytes: u16) Error!void { + const base: u17 = fbr_start * @as(u17, func); + try cmd52Write(0, base + cccr.blksize_l, @truncate(bytes)); + try cmd52Write(0, base + cccr.blksize_h, @truncate(bytes >> 8)); +} + +/// A bounded busy-wait, for the two places the SDIO specification asks for a delay between +/// retries. Same conservative frequency assumption as `Deadline`, in the same safe direction: on +/// this 90 MHz die a 10 ms request takes about 44 ms. +fn spinMicros(us: u32) void { + const d = Deadline.init(us); + while (!d.expired()) {} +} + +// ------------------------------------------------------------------------------------- CMD52 + +/// Read one byte from the card's register space. +/// +/// This is the whole minimal milestone: after `init` and `cardInit`, `cmd52Read(0, 0x00)` reads +/// CCCR offset 0 and the byte that comes back is the C6 answering. +pub fn cmd52Read(func: u3, addr: u17) Error!u8 { + const r = try sendCommand(.{ + .index = cmd_io_rw_direct, + .response = .short, + .check_crc = true, + }, cmd52Arg(false, func, addr, false, 0)); + return checkR5(r); +} + +/// Write one byte. The RAW flag is not set, matching `sdmmc_io_rw_direct` with SD_ARG_CMD52_WRITE +/// alone (`sdmmc_io.c:187`); `sdmmc_io_write_byte` adds SD_ARG_CMD52_EXCHANGE when it wants the +/// previous value back, which no caller here does. +pub fn cmd52Write(func: u3, addr: u17, value: u8) Error!void { + const r = try sendCommand(.{ + .index = cmd_io_rw_direct, + .response = .short, + .check_crc = true, + }, cmd52Arg(true, func, addr, false, value)); + _ = try checkR5(r); +} + +// ------------------------------------------------------------------------------------- CMD53 + +/// How one CMD53 is split. Two rules decide it, and both come from ESP-IDF rather than from the +/// SDIO specification, because both are properties of this controller: +/// +/// * **Block mode when the length is a whole number of 512-byte blocks**, byte mode otherwise. +/// In byte mode the count field is bytes and 0 encodes 512 ("See 5.3.1 SDIO simplified spec", +/// `sdmmc_io.c:351-355`), so one byte-mode command reaches 512 bytes and no further. +/// * **A byte-mode length of 4 or more must be a multiple of 4.** `sd_trans_sdmmc.c:526-532` +/// rejects anything else outright, and `sdmmc_io_read_bytes` works around it by splitting: +/// "host quirk: SDIO transfer with length not divisible by 4 bytes has to be split into two +/// transfers: one with aligned length, the other one for the remaining 1-3 bytes" +/// (`sdmmc_io.c:400-419`). So 6 bytes is two commands, 4 then 2, and 3 bytes is one. +/// +/// A caller that wants the split to be explicit - ESP-Hosted's block path does, because its +/// addresses increment across the split - can hand over one whole-block chunk at a time and get +/// exactly one block-mode command per call. A caller that does not can hand over any length. +const Chunk = struct { + block_mode: bool, + /// Bytes in this command. + len: u32, + /// The CMD53 count field: blocks in block mode, bytes in byte mode with 0 meaning 512. + count: u9, +}; + +fn nextChunk(remaining: u32) Chunk { + if (remaining >= io_block_size and remaining % io_block_size == 0) { + const max_blocks = bounce_len / io_block_size; + var blocks = remaining / io_block_size; + if (blocks > max_blocks) blocks = max_blocks; + return .{ + .block_mode = true, + .len = blocks * io_block_size, + .count = @intCast(blocks), + }; + } + var len = remaining; + if (len > io_block_size) len = io_block_size; + // The 4-byte rule. Below 4 bytes the whole request goes in one command; at or above it, the + // aligned part goes first and the 1-3 byte tail becomes the next chunk. + if (len >= 4 and len % 4 != 0) len &= ~@as(u32, 3); + return .{ + .block_mode = false, + .len = len, + .count = if (len == io_block_size) 0 else @intCast(len), + }; +} + +pub fn cmd53Read(func: u3, addr: u17, buf: []u8, incrementing: bool) Error!void { + var offset: u32 = 0; + var a: u32 = addr; + while (offset < buf.len) { + const c = nextChunk(@intCast(buf.len - offset)); + const arg = cmd53Arg(false, func, @truncate(a), c.block_mode, incrementing, c.count); + try dataTransfer(.read, arg, c.len, if (c.block_mode) io_block_size else c.len); + const dst = buf[offset..][0..c.len]; + const src = bufNc(); + for (dst, 0..) |*b, i| b.* = src[i]; + offset += c.len; + if (incrementing) a += c.len; + } +} + +pub fn cmd53Write(func: u3, addr: u17, data: []const u8, incrementing: bool) Error!void { + var offset: u32 = 0; + var a: u32 = addr; + while (offset < data.len) { + const c = nextChunk(@intCast(data.len - offset)); + const src = data[offset..][0..c.len]; + const dst = bufNc(); + for (src, 0..) |b, i| dst[i] = b; + // The IDMAC moves whole words, so a length that is not a multiple of 4 is rounded up + // (`sd_trans_sdmmc.c:127`). Zero the pad rather than send whatever the last transfer left. + var pad = c.len; + while (pad % 4 != 0) : (pad += 1) dst[pad] = 0; + const arg = cmd53Arg(true, func, @truncate(a), c.block_mode, incrementing, c.count); + try dataTransfer(.write, arg, c.len, if (c.block_mode) io_block_size else c.len); + offset += c.len; + if (incrementing) a += c.len; + } +} + +/// One CMD53 with its data phase, through the IDMAC and the bounce buffer. +/// +/// Order is ESP-IDF's (`sd_trans_sdmmc.c:524-568`): descriptor and transfer registers first, then +/// the command word, then wait for command-done and data-transfer-over in that order. Preparing +/// the DMA after starting the command would be a race against a card that answers immediately. +fn dataTransfer(dir: Direction, arg: u32, len: u32, blk: u32) Error!void { + std.debug.assert(len <= bounce_len); + + // The card must not still be holding DAT0 low from a previous write. + const busy = Deadline.init(busy_timeout_us); + while (status.get(data_busy) != 0) { + if (busy.expired()) return error.Busy; + } + + // As in `sendCommand`: the whole event set except this slot's card interrupt. `frun` is in + // `Event.data_errors` and nothing else ever clears it. + clearNonSlaveInterrupts(); + idsts.writeRaw(idsts_event_mask); + + const padded = (len + 3) & ~@as(u32, 3); + const d = descNc(); + d.buffer1 = busAddr(&dma.buf); + d.next = 0; + d.sizes = padded; // buffer1_size is [12:0]; buffer2 is unused + d.flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor | + Descriptor.last_descriptor | Descriptor.second_address_chained; + + setDataTransferLen(len); + setBlockSize(blk); + setDescriptorAddr(busAddr(&dma.desc)); + + // `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`, then the poll demand that tells the IDMAC to + // re-read a descriptor it may have parked on. + setDmaEnabled(true); + pldmnd.writeRaw(1); + + try startCommand(commandWord(.{ + .index = cmd_io_rw_extended, + .response = .short, + .check_crc = true, + .data = dir, + .slot = state.slot, + }), arg); + + _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us); + _ = try checkR5(resp0.raw()); + _ = try waitEvents(Event.dto, Event.data_errors, data_done_timeout_us); +} + +// -------------------------------------------------------------------- SDIO card interrupt (D1) + +/// Has the card asserted its interrupt line? +/// +/// Non-blocking, no side effect, straight out of RINTSTS bit 16+slot (`sdmmc_reg.h:621-631`). It +/// does **not** clear the bit; `clearSlaveInterrupt` does, deliberately, once a caller has decided +/// to act on it. +/// +/// Two different trigger behaviours meet at this bit and it is worth keeping them apart, because +/// conflating them sends you tuning the wrong knob: +/// +/// * **Card -> controller is an edge.** ESP-IDF: "SDIO interrupts are negedge sensitive ones: +/// the status bit is only set when first interrupt triggered" (`sd_host_sdmmc.c:396-402`). +/// That is why a waiter must check D1's level once before sleeping - an edge that arrived +/// while it was awake is not re-delivered. +/// * **Controller -> CLIC is a level.** RINTSTS is a sticky write-1-to-clear latch, so the +/// controller's output line stays asserted until software clears the bit that raised it. The +/// CLIC line therefore wants `.level`, and an edge trigger there would only hide a handler +/// that fails to deassert rather than fix it. +/// +/// This is the polling half. The interrupt half is a CLIC line and belongs to whoever owns the +/// scheduler: route `interrupt_source` with `hal.intr`, and in the handler mask the bit +/// (`setSlaveInterruptEnabled(false)`) before waking anybody. ESP-IDF does exactly that +/// (`sd_host_sdmmc.c:826-830`) and explains why at `:396-402`: "SDIO interrupts are negedge +/// sensitive ones: the status bit is only set when first interrupt triggered", so a handler that +/// leaves the bit unmasked and unhandled re-enters forever, and a waiter that sleeps without first +/// checking D1's level loses an edge that arrived while it was awake. +pub fn slaveInterruptPending() bool { + return rintsts.raw() & (Event.io_slot0 << state.slot) != 0; +} + +/// The raw masked-interrupt status word. Diagnostics only: a hang waiting on the card interrupt is +/// otherwise indistinguishable from a card that never asserted, and this is the register that tells +/// them apart. +pub fn interruptStatusRaw() u32 { + return rintsts.raw(); +} + +pub fn clearSlaveInterrupt() void { + rintsts.writeRaw(Event.io_slot0 << state.slot); +} + +/// INTMASK as written. The other half of "why is this line asserted": the controller's output is +/// RINTSTS AND INTMASK, and a diagnostic that prints only RINTSTS shows half the conjunction. +/// +/// **Not evidence about an arm.** INTMASK is an ordinary read/write register, so this returns +/// whatever the last store left - and on the interrupt path the last store is usually the *disarm*. +/// A caller that wants to know whether unmasking took effect must read the register back inside the +/// same masked region as the store; that is `armSlaveInterrupt`, and it exists because this +/// function was read as if it answered that question and it never could. +pub fn interruptMaskRaw() u32 { + return intmask.raw(); +} + +/// IDSTS - the IDMAC's own status word, which reaches the controller's interrupt output through +/// IDINTEN and *not* through INTMASK. +/// +/// The second independent reason the line can be asserted, and therefore the first thing to read +/// when a handler entry cannot be explained by RINTSTS. `muteInterrupts` leaves IDINTEN at zero so +/// this cannot raise the line in this driver; a foreign handler entry with bits set here means +/// something put IDINTEN back. +pub fn dmaStatusRaw() u32 { + return idsts.raw(); +} + +/// Clear every latched event *except* this slot's SDIO card interrupt. +/// +/// For an interrupt handler that has to lower the controller's output line without racing the +/// card: the card interrupt is the one event the handler is being woken for, and clearing it here +/// would drop the wakeup. Everything else - a stale command-done, a latched card-detect, an error +/// from a transfer that has already been reported - is safe to drop on the floor, and leaving any +/// of it latched while unmasked keeps the CLIC line high. +pub fn clearNonSlaveInterrupts() void { + rintsts.writeRaw(0x0003_ffff & ~(Event.io_slot0 << state.slot)); +} + +/// Unmask this slot's SDIO card interrupt, i.e. let it - and after `muteInterrupts`, *only* it - +/// reach the CLIC. RINTSTS records the event either way, so polling works without this. +/// +/// This is the whole of the masking half of arming a waiter: `muteInterrupts` has already left +/// every other bit of INTMASK and all of IDINTEN at zero, so `true` here makes this slot's card +/// interrupt the single reason the controller's output can assert - which is what a +/// level-triggered CLIC line with a one-cause handler requires. +/// +/// The *order* around it is the part that is easy to get wrong, and it belongs to whoever owns the +/// scheduler rather than here. `sd_host_slot_sdmmc_io_int_wait` (`sd_host_sdmmc.c:404-426`) is the +/// reference, and it is four steps: +/// +/// 1. `setSlaveInterruptEnabled(false)` - mask, so nothing arrives while the state is in flux. +/// 2. `clearSlaveInterrupt()` - drop the latched edge, so a stale one is not delivered as news. +/// 3. `slaveInterruptAsserted()` - **if true, act now and do not sleep.** The capture is a +/// negedge, so with D1 already low step 2 has just thrown away the only edge there will be. +/// 4. `setSlaveInterruptEnabled(true)` - unmask, and not before. Nothing can be lost between 2 +/// and 4: D1 is a level, and unmasking a bit RINTSTS has already latched asserts the line at +/// once. +/// +/// A handler on that line must mask again as its **unconditional first act**, on every path +/// including the one where the cause turns out not to be its own. The line is a level and it does +/// not lower itself. +pub fn setSlaveInterruptEnabled(on: bool) void { + // Masked, and that is not decoration. `sdioDispatch` performs *this same* read-modify-write on + // *this same* two-bit field, from an interrupt handler, as its unconditional first act. A task + // interrupted between the load and the store puts back the bit the handler had just cleared - + // re-arming a level-triggered line with nobody left waiting on it, which is precisely how the + // storm gets its second chance. Two CSR instructions, and `clkrst.Guard` composes: called from + // inside a handler, where MIE is already clear, it leaves MIE clear. + // + // The register writes are unchanged, so the `sdio_interrupt` differential case still compares + // the same resulting word against `sdmmc_ll_enable_sdio_interrupt`'s. + const guard = intr.mask(); + defer guard.release(); + const m = slotBit(); + const cur = intmask.get(sdio_int_mask); + intmask.modify(.{sdio_int_mask.is(if (on) cur | m else cur & ~m)}); +} + +/// What the controller reported the instant after this slot's card interrupt was unmasked. +pub const Armed = struct { + /// The bit the store was trying to set, i.e. `slaveInterruptMask()`. + want: u32, + /// INTMASK, read back inside the same masked region as the store. + intmask: u32, + /// MINTSTS - `RINTSTS & INTMASK`, and the only word the controller's output follows. Zero here + /// with `stuck()` true is the normal way to enter a sleep: the mask took and nothing is latched + /// yet. + mintsts: u32, + /// RINTSTS, for the case where the edge landed between the unmask and the read-back. + rintsts: u32, + + /// Did the unmask take effect? + pub inline fn stuck(self: Armed) bool { + return self.intmask & self.want != 0; + } +}; + +/// Unmask this slot's card interrupt and read the result back, both inside one masked region. +/// +/// This exists because the opposite conclusion was drawn from diagnostics that could not support +/// it. Every arming window on this board printed `intmask=0x00000000` and that was read as "the +/// unmask does not stick" - but `MARK PORT_SDIO_LAPSE` prints *after* the disarm, which had just +/// written that zero deliberately, and `interruptDiagnostics` runs on the application task, which +/// is never inside an arming window. Neither reading could ever have shown anything else, whatever +/// the hardware did. +/// +/// So the claim gets an instrument instead of an argument. Nothing runs between the store and the +/// three loads: no task, because the runtime is cooperative, and no handler, because MIE is clear. +/// A `stuck()` of false here is a fact about this register on this die; `stuck()` true retires the +/// hypothesis. +pub fn armSlaveInterrupt() Armed { + const guard = intr.mask(); + defer guard.release(); + const m = slotBit(); + intmask.modify(.{sdio_int_mask.is(intmask.get(sdio_int_mask) | m)}); + return .{ + .want = slaveInterruptMask(), + .intmask = intmask.raw(), + .mintsts = mintsts.raw(), + .rintsts = rintsts.raw(), + }; +} + +/// This slot's bit in RINTSTS/INTMASK/MINTSTS - the only interrupt cause a waiter here understands. +pub fn slaveInterruptMask() u32 { + return Event.io_slot0 << state.slot; +} + +/// Is the card asserting its interrupt *right now*? +/// +/// Read from D1's pad rather than from RINTSTS, because the two answer different questions: the +/// register says "a negedge was latched and not yet cleared", the pad says "the card is holding the +/// line low". Only the second is safe to test before sleeping, and it is what ESP-IDF tests - +/// `gpio_get_level(slot_ctx->io_config.d1_io) == 0` at `sd_host_sdmmc.c:413-415`. +/// +/// Requires D1's input buffer and matrix input to be configured, which `configurePins` does. +pub fn slaveInterruptAsserted() bool { + return gpio.getLevel(state.pins.d1) == 0; +} + +/// The masked status word - what the controller's interrupt output is actually looking at. +/// +/// Non-zero here and a silent CLIC means the delivery path above the controller is broken (source +/// routing, line enable, priority, threshold, mstatus.MIE). Zero here while `interruptStatusRaw` +/// is non-zero means the event is latched but masked, which is the normal resting state of this +/// driver. +pub fn interruptStatusMasked() u32 { + return mintsts.raw(); +} + +// ------------------------------------------------------------------------------- diagnostics + +/// `ets_printf` from the mask ROM, the same declaration `src/net/port.zig:75` makes and for the +/// same reason: this file's only module imports are `regs`, `mmio` and its sibling HALs, and the +/// symbol comes from the generated linker script rather than from any of them. +extern fn ets_printf(fmt: [*:0]const u8, ...) c_int; + +fn note(comptime fmt: [*:0]const u8, args: anytype) void { + _ = @call(.auto, ets_printf, .{fmt} ++ args); +} + +inline fn yesno(b: bool) u32 { + return @intFromBool(b); +} + +/// Print the whole card-interrupt delivery chain, in the order a signal traverses it, so that one +/// flash says which link is broken. Reads registers only: no loop, no wait, no side effect on any +/// of the state it reports. +/// +/// The chain has four links and each line below covers one: +/// +/// * `SDIO_DIAG_PAD` - the card's end. `asserted=1` means D1 is low, i.e. the C6 is requesting +/// service at this instant. `ie=0` or `in_src` not equal to D1's pad number means the +/// controller cannot see D1 at all, and no amount of unmasking will help. +/// * `SDIO_DIAG_TIEOFF` - the two matrix inputs with no pin on this board. `card_int_n` must read +/// 63 (constant one, inactive) and `card_detect_n` 62 (constant zero, card present), both with +/// `from_matrix=1`. A `card_int_n` stuck at a constant *zero* is an interrupt that is asserted +/// before software ever runs, so the first negedge happens before anyone is watching and no +/// second one ever comes. +/// * `SDIO_DIAG_CTLR` - the controller's end. `latched` is RINTSTS's bit for this slot, +/// `unmasked` is INTMASK's, and `mintsts` is the conjunction the interrupt output follows. +/// `idsts`/`idinten` are the other, independent reason this output can be asserted. +/// +/// **`unmasked=0` here is the resting state and is not a finding.** This function is called +/// from an application task; the card interrupt is unmasked only inside an arming window, on +/// the transport's own task, and is masked again by the handler or the disarm before that task +/// yields. So an application can never observe the mask up, whatever the hardware does, and +/// reading a zero here as "the unmask does not stick" is what cost this path a week. The +/// register read that can answer that question is `armSlaveInterrupt`. +/// * `SDIO_DIAG_CLIC` - delivery. An unrouted source, a clear `enabled`, a priority at or below +/// `thresh`, or `mie=0` each mean the line exists and cannot arrive. +pub fn interruptDiagnostics() void { + const m = slaveInterruptMask(); + const rsts = rintsts.raw(); + const imask = intmask.raw(); + const d1 = state.pins.d1; + const d1_in = gpio.matrixInSource(sig.cdata1); + const ci = gpio.matrixInSource(sig.card_int); + const cdet = gpio.matrixInSource(sig.card_detect); + + note("MARK SDIO_DIAG_PAD d1=gpio%u level=%u asserted=%u ie=%u in_src=%u from_matrix=%u inv=%u expect in_src=%u\r\n", .{ + @as(u32, d1), + @as(u32, gpio.getLevel(d1)), + yesno(slaveInterruptAsserted()), + yesno(gpio.isInputEnabled(d1)), + @as(u32, d1_in.pin), + yesno(d1_in.from_matrix), + yesno(d1_in.inverted), + @as(u32, d1), + }); + note("MARK SDIO_DIAG_TIEOFF card_int_n=%u/%u card_detect_n=%u/%u expect 63/1 and 62/1\r\n", .{ + @as(u32, ci.pin), yesno(ci.from_matrix), + @as(u32, cdet.pin), yesno(cdet.from_matrix), + }); + note("MARK SDIO_DIAG_CTLR slot=%u bit=0x%05x latched=%u unmasked=%u rintsts=0x%08x intmask=0x%08x mintsts=0x%08x\r\n", .{ + @as(u32, state.slot), + m, + yesno(rsts & m != 0), + yesno(imask & m != 0), + rsts, + imask, + interruptStatusMasked(), + }); + note("MARK SDIO_DIAG_CTRL ctrl=0x%08x int_enable=%u idsts=0x%08x idinten=0x%08x status=0x%08x clkena=0x%08x\r\n", .{ + ctrl.raw(), + ctrl.get(int_enable), + idsts.raw(), + idinten.raw(), + status.raw(), + clkena.raw(), + }); + + const src: u32 = @intFromEnum(interrupt_source); + if (intr.routedLine(interrupt_source)) |line| { + note("MARK SDIO_DIAG_CLIC source=%u line=%u enabled=%u pending=%u trigger=%u prio=%u thresh=%u mie=%u\r\n", .{ + src, + @as(u32, line), + yesno(intr.isEnabled(line)), + yesno(intr.isPending(line)), + @as(u32, @intFromEnum(intr.getTrigger(line))), + @as(u32, intr.getPriority(line)), + @as(u32, intr.getThreshold()), + yesno(intr.globalEnabled()), + }); + } else { + note("MARK SDIO_DIAG_CLIC source=%u UNROUTED - no CLIC line can deliver this interrupt\r\n", .{src}); + } +} + +// --------------------------------------------------------------------------------- host tests +// +// Everything below runs on the host under `zig build test`. It covers the two things in this file +// that are pure functions of their arguments - the command word and the CMD52/CMD53 argument +// layouts - plus the divider table and the chunking rule. The register sequences are not testable +// here; that is what src/oracle/sdmmc_cases.zig is for. + +const testing = std.testing; + +test "CMD52 read is the word ESP-IDF builds" { + // make_hw_cmd for {opcode 52, SCF_CMD_AC | SCF_RSP_R5}: response_expect (R5 is PRESENT), + // check_response_crc (R5 has CRC), wait_complete (not CMD0/12/11), no data. Then + // sd_host_slot_start_command adds use_hold_reg, card_num and start_command. + const w = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_2174), w); +} + +test "CMD52 on slot 0 differs from slot 1 only in card_num" { + const s0 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 0 }); + const s1 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA000_2174), s0); + try testing.expectEqual(@as(u32, 1 << 16), s0 ^ s1); +} + +test "CMD53 sets data_expected, and rw only when writing" { + const rd = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .read, .slot = 1 }); + const wr = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .write, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_2375), rd); + try testing.expectEqual(@as(u32, 0xA001_2775), wr); + try testing.expectEqual(@as(u32, 1 << 10), rd ^ wr); +} + +test "CMD0 sends the init sequence and expects nothing back" { + // The only command make_hw_cmd gives send_init and denies wait_complete. + const w = commandWord(.{ .index = 0, .send_init = true, .wait_prvdata = false, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_8000), w); + try testing.expectEqual(@as(u32, 0), w & (1 << 6)); // no response expected +} + +test "CMD5's response CRC is not checked" { + // R4 is SCF_RSP_PRESENT alone (sd_protocol_types.h:141) - the OCR response carries no valid + // CRC7, and checking it would fail every card. + const w = commandWord(.{ .index = 5, .response = .short, .check_crc = false, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA001_2045), w); + try testing.expectEqual(@as(u32, 0), w & (1 << 8)); +} + +test "CMD3 and CMD7 do check it" { + try testing.expectEqual( + @as(u32, 0xA001_2143), + commandWord(.{ .index = 3, .response = .short, .check_crc = true, .slot = 1 }), + ); + try testing.expectEqual( + @as(u32, 0xA001_2147), + commandWord(.{ .index = 7, .response = .short, .check_crc = true, .slot = 1 }), + ); +} + +test "the clock update command sends nothing to the card" { + const w = commandWord(.{ .index = 0, .update_clock = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 0xA021_2000), w); + try testing.expectEqual(@as(u32, 1 << 21), w & (1 << 21)); + try testing.expectEqual(@as(u32, 0), w & (1 << 6)); +} + +test "a long response sets response_length as well as response_expect" { + const w = commandWord(.{ .index = 2, .response = .long, .check_crc = true, .slot = 1 }); + try testing.expectEqual(@as(u32, 1 << 7), w & (1 << 7)); + try testing.expectEqual(@as(u32, 1 << 6), w & (1 << 6)); +} + +test "every command word starts the command and uses the hold register" { + for ([_]Command{ + .{ .index = 52, .response = .short, .check_crc = true }, + .{ .index = 53, .response = .short, .check_crc = true, .data = .read }, + .{ .index = 0, .send_init = true, .wait_prvdata = false }, + }) |c| { + const w = commandWord(c); + try testing.expect(w & (1 << 31) != 0); + try testing.expect(w & (1 << 29) != 0); + } +} + +test "CMD52 argument layout" { + // Read CCCR 0x00 on function 0: everything zero. + try testing.expectEqual(@as(u32, 0), cmd52Arg(false, 0, 0x00, false, 0)); + // Write 0x02 to CCCR 0x02 (I/O enable) on function 0. + try testing.expectEqual(@as(u32, 0x8000_0402), cmd52Arg(true, 0, 0x02, false, 0x02)); + // Function 1, address 0x1F800, data 0xAB, with the read-after-write flag. + const a = cmd52Arg(true, 1, 0x1F800, true, 0xAB); + try testing.expectEqual(@as(u32, 1), a >> 31); + try testing.expectEqual(@as(u32, 1), (a >> 28) & 0x7); + try testing.expectEqual(@as(u32, 1), (a >> 27) & 1); + try testing.expectEqual(@as(u32, 0x1F800), (a >> 9) & 0x1FFFF); + try testing.expectEqual(@as(u32, 0xAB), a & 0xFF); +} + +test "CMD53 argument layout, both modes" { + // Block mode, function 1, address 0, incrementing, one block. + const blk = cmd53Arg(false, 1, 0, true, true, 1); + try testing.expectEqual(@as(u32, 0x1C00_0001), blk); + // Byte mode, function 1, fixed address, 12 bytes - the ESP-Hosted length read. + const byt = cmd53Arg(false, 1, 0x058, false, false, 12); + try testing.expectEqual(@as(u32, 0x1000_B00C), byt); + // Writing sets bit 31 and nothing else. + try testing.expectEqual( + @as(u32, 1) << 31, + cmd53Arg(true, 1, 0x058, false, false, 12) ^ byt, + ); +} + +test "byte mode encodes 512 as a count of zero" { + // SDIO simplified spec 5.3.1, as applied at sdmmc_io.c:351-355. The chunker never produces + // this case - a 512-byte request is a whole block and goes block mode - so the encoder is + // checked directly. ESP-Hosted can still reach it: a 512-byte read at a fixed address. + try testing.expectEqual(@as(u32, 0), cmd53Arg(false, 1, 0, false, true, 0) & 0x1ff); + const c = nextChunk(512); + try testing.expect(c.block_mode); + try testing.expectEqual(@as(u32, 512), c.len); + try testing.expectEqual(@as(u9, 1), c.count); +} + +test "chunking splits on block boundaries and clamps to the bounce buffer" { + // Whole blocks, within the buffer: one block-mode command. + try testing.expectEqual(@as(u32, 1024), nextChunk(1024).len); + try testing.expect(nextChunk(1024).block_mode); + // More blocks than fit: clamped to bounce_len, still block mode, still whole blocks. + const big = nextChunk(8192); + try testing.expect(big.block_mode); + try testing.expectEqual(bounce_len, big.len); + try testing.expectEqual(@as(u9, bounce_len / 512), big.count); + // Not a block multiple: byte mode, count in bytes. + const odd = nextChunk(12); + try testing.expect(!odd.block_mode); + try testing.expectEqual(@as(u32, 12), odd.len); + try testing.expectEqual(@as(u9, 12), odd.count); + // Longer than one byte-mode command can carry: clamped to 512. + const long = nextChunk(1000); + try testing.expect(!long.block_mode); + try testing.expectEqual(@as(u32, 512), long.len); +} + +test "the controller's 4-byte rule turns a 6-byte transfer into 4 then 2" { + // sd_trans_sdmmc.c:526-532 rejects a length that is >= 4 and not a multiple of 4 outright. + const first = nextChunk(6); + try testing.expect(!first.block_mode); + try testing.expectEqual(@as(u32, 4), first.len); + const second = nextChunk(6 - first.len); + try testing.expectEqual(@as(u32, 2), second.len); + // Under four bytes the whole thing goes in one command; that is the case the rule exempts. + try testing.expectEqual(@as(u32, 3), nextChunk(3).len); + try testing.expectEqual(@as(u32, 1), nextChunk(1).len); + // And every chunk a loop produces is either aligned or a final short tail. + var remaining: u32 = 1023; + var commands: u32 = 0; + while (remaining > 0) { + const c = nextChunk(remaining); + try testing.expect(c.len > 0); + try testing.expect(c.len < 4 or c.len % 4 == 0); + remaining -= c.len; + commands += 1; + try testing.expect(commands < 8); // 512 + 508 + 3, not an unbounded walk + } +} + +test "divider table reproduces ESP-IDF's three named frequencies" { + try testing.expectEqual(Dividers{ .host = 10, .card = 20 }, dividersFor(400)); + try testing.expectEqual(Dividers{ .host = 8, .card = 0 }, dividersFor(20_000)); + try testing.expectEqual(Dividers{ .host = 4, .card = 0 }, dividersFor(40_000)); +} + +test "divider table lands on or below the requested frequency" { + for ([_]u32{ 400, 1_000, 5_000, 10_000, 20_000, 25_000, 40_000 }) |khz| { + const d = dividersFor(khz); + const div: u64 = @as(u64, d.host) * (if (d.card == 0) @as(u64, 1) else @as(u64, d.card) * 2); + const actual_khz = 160_000 / div; + try testing.expect(actual_khz <= khz); + } +} + +test "the DMA region is one aligned block of exactly the documented size" { + try testing.expectEqual(@as(usize, 16), @sizeOf(Descriptor)); + try testing.expectEqual(@as(usize, 64 + bounce_len), @sizeOf(DmaRegion)); + try testing.expectEqual(@as(usize, 0), @offsetOf(DmaRegion, "desc")); + try testing.expectEqual(@as(usize, 64), @offsetOf(DmaRegion, "buf")); + // Both halves of what the one-shot cache maintenance call needs: a base on a cache line and a + // length that is a whole number of them. The alignment is on the variable, not on the type - + // `@alignOf(DmaRegion)` is 4 - so it has to be checked on the object. + try testing.expectEqual(@as(usize, 0), @intFromPtr(&dma) % cache_line); + try testing.expectEqual(@as(usize, 0), @sizeOf(DmaRegion) % cache_line); +} + +test "the non-cacheable alias is a fixed offset and nothing more" { + var cell: u32 = 0; + const a = cachedAddr(&cell); + try testing.expectEqual(a +% @as(usize, 0x4000_0000), uncachedAddr(&cell)); + // And it is the offset ESP-IDF uses, not one this file invented. + try testing.expectEqual(@as(u32, 0x4000_0000), non_cacheable_offset); +} + +test "descriptor flags are the bits sdmmc_struct.h names" { + try testing.expectEqual(@as(u32, 1 << 2), Descriptor.last_descriptor); + try testing.expectEqual(@as(u32, 1 << 3), Descriptor.first_descriptor); + try testing.expectEqual(@as(u32, 1 << 4), Descriptor.second_address_chained); + try testing.expectEqual(@as(u32, 1 << 31), Descriptor.owned_by_idmac); + // The word a single-descriptor transfer writes. + const flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor | + Descriptor.last_descriptor | Descriptor.second_address_chained; + try testing.expectEqual(@as(u32, 0x8000_001C), flags); +} + +test "the default interrupt mask is ESP-IDF's SDMMC_LL_EVENT_DEFAULT" { + // sdmmc_ll.h:64-69, expanded: CD|RESP_ERR|CMD_DONE|DATA_OVER|RCRC|DCRC|RTO|DTO|HTO|HLE|SBE|EBE + try testing.expectEqual(@as(u32, 0xB7CF), Event.default); + // and it deliberately excludes the two per-FIFO-word requests and both SDIO card interrupts. + try testing.expectEqual(@as(u32, 0), Event.default & (Event.txdr | Event.rxdr)); + try testing.expectEqual(@as(u32, 0), Event.default & (Event.io_slot0 | Event.io_slot1)); +} + +test "the armed mask drops card detect, and nothing else" { + // The bit that produced an unstoppable CLIC line 21: cd latches during pin setup, nothing in + // the command path clears it, and while it is unmasked the controller's output never + // deasserts. `configureInterrupts` writes `armed`, not `default`. + try testing.expectEqual(@as(u32, 0xB7CE), Event.armed); + try testing.expectEqual(@as(u32, 0), Event.armed & Event.cd); + try testing.expectEqual(Event.cd, Event.default ^ Event.armed); + // Every event a transfer actually waits on survives the change. + for ([_]u32{ Event.cmd_done, Event.dto, Event.re, Event.rcrc, Event.dcrc, Event.rto, Event.drto, Event.hto, Event.hle, Event.sbe, Event.ebe }) |e| { + try testing.expect(Event.armed & e != 0); + } +} diff --git a/src/hal/systimer.zig b/src/hal/systimer.zig new file mode 100644 index 0000000..dc316bf --- /dev/null +++ b/src/hal/systimer.zig @@ -0,0 +1,151 @@ +//! SYSTIMER: two 52-bit counters on a fixed clock, plus three comparators each. +//! +//! This is the most useful peripheral on the chip for bring-up work and the cheapest to trust. Its +//! source is fixed - XTAL at 40 MHz, divided to 16 MHz (`clk_tree_defs.h:196-198`) - so unlike the +//! CPU cycle counter its rate does not move when the clock tree is reconfigured, and unlike the +//! timer groups it needs no divider arithmetic and no pads. +//! +//! Reading it is a **sequence**, not a load, and that is the interesting part: +//! +//! write UNIT0_UPDATE = 1 -> ask the peripheral to latch its counter +//! poll UNIT0_VALUE_VALID -> wait for the latch +//! read VALUE_HI, then VALUE_LO -> read the latched pair +//! +//! Skip the handshake and you read a value that is being incremented underneath you: the low word +//! can wrap between the two loads, so `hi` belongs to one instant and `lo` to the next, and the +//! result jumps backwards by 2^32 ticks about once every 268 seconds at 16 MHz. A register +//! snapshot taken after either version looks identical - which is exactly why the differential +//! harness records the *write trace* as well as the final state. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// Ticks per second. XTAL/2.5 = 16 MHz, fixed: `SYSTIMER_CLK_SRC_XTAL` with the divider ESP-IDF +/// programs in `systimer_hal_init`. Not derived from the CPU clock, which on this board is whatever +/// the bootloader left (measured ~90 MHz, not the 360 the part is rated for). +pub const hz: u32 = 16_000_000; + +const conf = Reg.at(regs.SYSTIMER_CONF_REG); +const clk_en = Field.of(regs.SYSTIMER_CLK_EN_S, regs.SYSTIMER_CLK_EN_V); + +/// The two counter units. `unit_op` holds the update/valid handshake bits, `value_hi`/`value_lo` the +/// latched result. Strides are derived from consecutive macros, not assumed. +const unit_op = mmio.RegArray(regs.SYSTIMER_UNIT0_OP_REG, regs.SYSTIMER_UNIT1_OP_REG, 2); +const unit_value_hi = mmio.RegArray(regs.SYSTIMER_UNIT0_VALUE_HI_REG, regs.SYSTIMER_UNIT1_VALUE_HI_REG, 2); +const unit_value_lo = mmio.RegArray(regs.SYSTIMER_UNIT0_VALUE_LO_REG, regs.SYSTIMER_UNIT1_VALUE_LO_REG, 2); + +// The per-unit fields split into two groups, and the split is not obvious from the names. +// +// `update` and `valid` live in a *per-unit* register (UNIT0_OP_REG, UNIT1_OP_REG) and therefore sit +// at the same bit in each - asserted below, so indexing the register is enough. +// +// `work_en` is different: both units' enables live in the *shared* SYSTIMER_CONF_REG, at bits 30 and +// 29 respectively. A first draft of this file used unit 0's field for both, which would have enabled +// the wrong counter and left the requested one dead; the comptime assert caught it before it ever +// reached the chip. Hence a per-unit lookup rather than one constant. +const update = Field.of(regs.SYSTIMER_TIMER_UNIT0_UPDATE_S, regs.SYSTIMER_TIMER_UNIT0_UPDATE_V); +const valid = Field.of(regs.SYSTIMER_TIMER_UNIT0_VALUE_VALID_S, regs.SYSTIMER_TIMER_UNIT0_VALUE_VALID_V); +const value_hi = Field.of(regs.SYSTIMER_TIMER_UNIT0_VALUE_HI_S, regs.SYSTIMER_TIMER_UNIT0_VALUE_HI_V); + +comptime { + const update1 = Field.of(regs.SYSTIMER_TIMER_UNIT1_UPDATE_S, regs.SYSTIMER_TIMER_UNIT1_UPDATE_V); + const valid1 = Field.of(regs.SYSTIMER_TIMER_UNIT1_VALUE_VALID_S, regs.SYSTIMER_TIMER_UNIT1_VALUE_VALID_V); + if (update1.shift != update.shift or valid1.shift != valid.shift) + @compileError("the systimer units' OP registers disagree on bit positions; index per unit"); + // The other half of the same story: these two MUST differ, because they share a register. + if (workEn(.unit0).shift == workEn(.unit1).shift) + @compileError("both work_en fields claim the same bit of SYSTIMER_CONF; one macro is wrong"); +} + +inline fn workEn(comptime unit: Unit) Field { + return switch (unit) { + .unit0 => Field.of(regs.SYSTIMER_TIMER_UNIT0_WORK_EN_S, regs.SYSTIMER_TIMER_UNIT0_WORK_EN_V), + .unit1 => Field.of(regs.SYSTIMER_TIMER_UNIT1_WORK_EN_S, regs.SYSTIMER_TIMER_UNIT1_WORK_EN_V), + }; +} + +pub const Unit = enum(u1) { unit0 = 0, unit1 = 1 }; + +/// The counter's own clock gate, inside the peripheral and separate from the bus clock gate in +/// HP_SYS_CLKRST. +pub fn setEnabled(on: bool) void { + conf.modify(.{clk_en.is(@intFromBool(on))}); +} + +pub fn setUnitEnabled(comptime unit: Unit, on: bool) void { + conf.modify(.{workEn(unit).is(@intFromBool(on))}); +} + +/// Bring the peripheral up: bus clock and reset through CLKRST, then its internal gate and unit. +/// +/// Deliberately does *not* reprogram the clock source or divider. The bootloader has already set +/// those, ESP-IDF's own `systimer_hal_init` would set them the same way, and re-running that on a +/// live counter makes the timebase jump - which would corrupt any measurement taken across the call. +pub fn init() void { + clkrst.setClockEnabled(.systimer, true); + setEnabled(true); + setUnitEnabled(.unit0, true); +} + +/// The 52-bit counter, latched through the update/valid handshake. +/// +/// Returns null if the peripheral does not acknowledge within `spins` reads, rather than spinning +/// forever: a systimer whose clock is gated off never sets `valid`, and hanging in a HAL call with +/// no output is the worst possible way to report that. +pub fn read(unit: Unit) ?u64 { + const i: u32 = @intFromEnum(unit); + const op = unit_op.at(i); + + // Ask for a snapshot, and clear the previous handshake in the same store. + // + // This has to be a read-modify-write, and a whole-word `write` is a bug. UPDATE is bit 30 and + // `WT`, so writing it as a single store looks right - but VALUE_VALID is bit 29 of the same word + // and is `R/SS/WTC`, write-1-to-clear (systimer_reg.h). A whole-word store writes 0 there, which + // is the no-op for a W1C bit, so the valid flag from the *previous* snapshot is never cleared: + // after one successful read it stays set forever, the poll below exits immediately on a stale + // flag, and the HI/LO pair that follows can straddle two different snapshots - precisely the + // tearing this handshake exists to prevent. + // + // ESP-IDF gets this right by accident of its idiom: `systimer_ll_counter_snapshot` assigns a + // bitfield of a `volatile` union, which compiles to a 32-bit read-modify-write that writes bit + // 29 back as 1 whenever it read 1, clearing it and re-arming in one store. This does the same + // thing deliberately. + // + // The register differential cannot see this: once any snapshot has completed, UNIT0_OP reads + // 0x2000_0000 under either version. + op.writeRaw(op.raw() | update.mask()); + + var spins: u32 = 0; + while (op.get(valid) == 0) { + spins += 1; + if (spins > 10_000) return null; + } + + // Order matters less than the latch does - both words are frozen now - but read high first to + // match ESP-IDF's LL, so the write/read trace lines up under differential test. + const hi: u64 = unit_value_hi.at(i).get(value_hi); + const lo: u64 = unit_value_lo.at(i).raw(); + return (hi << 32) | lo; +} + +/// Microseconds since the counter started, from the 16 MHz tick. +pub fn micros(unit: Unit) ?u64 { + const ticks = read(unit) orelse return null; + return ticks / (hz / 1_000_000); +} + +/// Busy-wait. Uses the counter rather than the CPU cycle count, so the delay is right regardless of +/// what the CPU clock happens to be. +pub fn delayMicros(us: u32) void { + const start = read(.unit0) orelse return; + const target = start + @as(u64, us) * (hz / 1_000_000); + while (true) { + const now = read(.unit0) orelse return; + if (now >= target) return; + } +} diff --git a/src/hal/timg.zig b/src/hal/timg.zig new file mode 100644 index 0000000..f669387 --- /dev/null +++ b/src/hal/timg.zig @@ -0,0 +1,513 @@ +//! The timer groups: TIMG0 and TIMG1, each two general-purpose 54-bit timers plus one MWDT. +//! +//! Three unrelated functions share one register block (timg_ll.h:7 says so in as many words): +//! the general-purpose timers, the main watchdog, and RTC clock calibration. Only the first two are +//! here; calibration belongs to the clock tree, and ETM and interrupts are deliberately absent. +//! +//! Four things about this block cost real care, all of them taken from ESP-IDF's LL rather than +//! guessed at: +//! +//! **Reading the counter is a sequence, not a load** (timer_ll.h:248-269). The counter lives in a +//! different clock domain from the register file, so its value only appears in `TxLO`/`TxHI` after a +//! software capture: +//! +//! write TIMG_TxUPDATE = 1 -> ask for a capture +//! poll until TIMG_Tx_UPDATE == 0 -> the hardware clears it when the pair is latched +//! read TxHI, then TxLO -> 22 bits + 32 bits = the 54-bit count +//! +//! Note the polarity: unlike SYSTIMER, which sets a separate `VALUE_VALID` bit, this peripheral +//! *clears the request bit* to acknowledge. Waiting for it to become 1 hangs forever; not waiting at +//! all returns whatever the last capture left, which for a never-captured timer is 0 and therefore +//! looks like a stopped timer rather than like a bug. +//! +//! **The watchdog registers are write-protected, and the key is the reset value** (mwdt_ll.h:231-244 +//! and timer_group_reg.h, TIMG_WDT_WKEY: "If the register contains a different value than its reset +//! value, write protection is enabled", default 1356348065 = 0x50D83AA1). So "unlock" means writing +//! the key back, and "lock" means writing anything else - IDF writes 0. A watchdog register write +//! made while locked is silently dropped, which is the failure mode this file's API shape exists to +//! prevent: every MWDT operation is a method on the `Watchdog` handle returned by `unlock`, and +//! there is no way to reach one without holding it: +//! +//! const wdt = timg.unlock(.timg1); +//! defer wdt.release(); +//! wdt.setStage(.stage0, 2_000_000, .reset_system); +//! +//! **Watchdog configuration is committed asynchronously.** Every write to WDTCONFIG0-5 has to be +//! followed by `WDT_CONF_UPDATE_EN` (mwdt_ll.h:122, and again after every other config write), which +//! is a write-to-trigger bit. The exception is `WDT_EN` itself: `mwdt_ll_enable`/`_disable` +//! (mwdt_ll.h:61-77) do *not* pulse it, so neither does `setEnabled` - matching IDF exactly matters +//! more here than consistency, because the differential harness compares the resulting word. +//! +//! **Do not resurrect a watchdog you are not feeding.** TIMG0 hosts MWDT0, which this image's +//! bootloader has already disabled, and `TIMG_WDT_FLASHBOOT_MOD_EN` defaults to 1 and runs the +//! watchdog *independently of* `WDT_EN` (mwdt_ll.h:186-188). Resetting a timer group therefore +//! re-arms flash-boot protection and reboots the board a moment later with nothing on the console to +//! explain it; `clkrst.resetPeripheral` clears the bit as part of the reset for exactly this reason +//! (its `clears_flashboot` flag), which is why nothing in this file pulses a reset bit itself. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// TIMG_LL_INST_NUM (timg_ll.h:20). +pub const group_count = 2; +/// TIMG_LL_GPTIMERS_PER_INST (timg_ll.h:23). Two per group on the P4, unlike the C-series parts. +pub const timers_per_group = 2; +/// TIMER_LL_COUNTER_BIT_WIDTH (timer_ll.h:25). 32 bits in `TxLO` plus 22 in `TxHI`. +pub const counter_bits = 54; + +pub const Group = enum(u1) { timg0 = 0, timg1 = 1 }; +pub const Timer = enum(u1) { t0 = 0, t1 = 1 }; + +// -------------------------------------------------------------------------------- addressing +// +// The macros are indexed two different ways at once and neither is derivable from the other: +// `TIMG_T0CONFIG_REG(i)` takes the *group*, while the *timer* is baked into the macro name +// (`T0CONFIG` vs `T1CONFIG`). Rather than duplicate every accessor per timer, the timer index is +// turned into a stride - but a stride assumed is a stride that eventually writes into the next +// register, so both strides are checked at comptime against the macros for the other instance. + +const group_stride = mmio.addr(regs.TIMG_T0CONFIG_REG(1)) - mmio.addr(regs.TIMG_T0CONFIG_REG(0)); +const timer_stride = mmio.addr(regs.TIMG_T1CONFIG_REG(0)) - mmio.addr(regs.TIMG_T0CONFIG_REG(0)); + +// Absolute addresses of group 0 / timer 0's registers. Every other (group, timer) is these plus a +// multiple of the two strides. +const a_config = mmio.addr(regs.TIMG_T0CONFIG_REG(0)); +const a_lo = mmio.addr(regs.TIMG_T0LO_REG(0)); +const a_hi = mmio.addr(regs.TIMG_T0HI_REG(0)); +const a_update = mmio.addr(regs.TIMG_T0UPDATE_REG(0)); +const a_alarm_lo = mmio.addr(regs.TIMG_T0ALARMLO_REG(0)); +const a_alarm_hi = mmio.addr(regs.TIMG_T0ALARMHI_REG(0)); +const a_load_lo = mmio.addr(regs.TIMG_T0LOADLO_REG(0)); +const a_load_hi = mmio.addr(regs.TIMG_T0LOADHI_REG(0)); +const a_load = mmio.addr(regs.TIMG_T0LOAD_REG(0)); + +comptime { + // The timer sub-block is contiguous and uniform - assert it, per register, rather than trust + // that 0x24 happens to be right for all nine. + const pairs = .{ + .{ a_config, mmio.addr(regs.TIMG_T1CONFIG_REG(0)) }, + .{ a_lo, mmio.addr(regs.TIMG_T1LO_REG(0)) }, + .{ a_hi, mmio.addr(regs.TIMG_T1HI_REG(0)) }, + .{ a_update, mmio.addr(regs.TIMG_T1UPDATE_REG(0)) }, + .{ a_alarm_lo, mmio.addr(regs.TIMG_T1ALARMLO_REG(0)) }, + .{ a_alarm_hi, mmio.addr(regs.TIMG_T1ALARMHI_REG(0)) }, + .{ a_load_lo, mmio.addr(regs.TIMG_T1LOADLO_REG(0)) }, + .{ a_load_hi, mmio.addr(regs.TIMG_T1LOADHI_REG(0)) }, + .{ a_load, mmio.addr(regs.TIMG_T1LOAD_REG(0)) }, + }; + for (pairs) |p| { + if (p[1] - p[0] != timer_stride) @compileError( + "the two timers' registers are not a uniform stride apart; index them per timer", + ); + } + // And the group stride is the same for a register other than CONFIG. + if (mmio.addr(regs.TIMG_T0LO_REG(1)) - a_lo != group_stride) + @compileError("the two timer groups are not a uniform stride apart"); + + // T0's and T1's *fields* sit at the same bit positions in their respective registers, which is + // what makes one set of Field constants enough. If a future register set moves one of them, + // this stops the build instead of writing the divider into the alarm enable. + const t1_divider = Field.of(regs.TIMG_T1_DIVIDER_S, regs.TIMG_T1_DIVIDER_V); + const t1_en = Field.of(regs.TIMG_T1_EN_S, regs.TIMG_T1_EN_V); + const t1_update = Field.of(regs.TIMG_T1_UPDATE_S, regs.TIMG_T1_UPDATE_V); + const t1_hi = Field.of(regs.TIMG_T1_HI_S, regs.TIMG_T1_HI_V); + if (t1_divider.shift != divider.shift or t1_divider.width != divider.width or + t1_en.shift != counter_en.shift or t1_update.shift != update.shift or + t1_hi.width != count_hi.width) + @compileError("timer 0 and timer 1 disagree on field positions; look up fields per timer"); +} + +inline fn tReg(comptime a0: u32, g: Group, t: Timer) Reg { + return Reg.atAddress(a0 + + group_stride * @as(u32, @intFromEnum(g)) + + timer_stride * @as(u32, @intFromEnum(t))); +} + +/// `a0` is not comptime: the stage-timeout registers are picked by a runtime `Stage` +/// (`stageHoldAddr`), and every other caller passes a constant that folds anyway. +inline fn gReg(a0: u32, g: Group) Reg { + return Reg.atAddress(a0 + group_stride * @as(u32, @intFromEnum(g))); +} + +// TxCONFIG fields. `divcnt_rst` is write-to-trigger; the rest are plain R/W. +const alarm_en = Field.of(regs.TIMG_T0_ALARM_EN_S, regs.TIMG_T0_ALARM_EN_V); +const divcnt_rst = Field.of(regs.TIMG_T0_DIVCNT_RST_S, regs.TIMG_T0_DIVCNT_RST_V); +const divider = Field.of(regs.TIMG_T0_DIVIDER_S, regs.TIMG_T0_DIVIDER_V); +const autoreload = Field.of(regs.TIMG_T0_AUTORELOAD_S, regs.TIMG_T0_AUTORELOAD_V); +const increase = Field.of(regs.TIMG_T0_INCREASE_S, regs.TIMG_T0_INCREASE_V); +const counter_en = Field.of(regs.TIMG_T0_EN_S, regs.TIMG_T0_EN_V); +const update = Field.of(regs.TIMG_T0_UPDATE_S, regs.TIMG_T0_UPDATE_V); +const count_hi = Field.of(regs.TIMG_T0_HI_S, regs.TIMG_T0_HI_V); +const alarm_value_hi = Field.of(regs.TIMG_T0_ALARM_HI_S, regs.TIMG_T0_ALARM_HI_V); +const load_value_hi = Field.of(regs.TIMG_T0_LOAD_HI_S, regs.TIMG_T0_LOAD_HI_V); + +// ------------------------------------------------------------------------------ timer clocks +// +// The timers' function clock is selected and gated in HP_SYS_CLKRST, not in the timer group: group 0 +// in PERI_CLK_CTRL20 and group 1 in PERI_CLK_CTRL21 (timer_ll.h:117-129, :146-160). Two shared +// registers, so both operations take the interrupt guard - the same read-modify-write hazard +// `clkrst` exists for. + +const peri_clk_ctrl20 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL20_REG); +const peri_clk_ctrl21 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL21_REG); + +/// The three function clocks a GP timer can run from, with the encodings from +/// `timer_ll_set_clock_source` (timer_ll.h:100-116). The numbering is not the enum order anyone +/// would pick: XTAL is 0, RC_FAST is 1, PLL_F80M is 2. +pub const ClockSource = enum(u2) { + xtal = 0, + rc_fast = 1, + pll_f80m = 2, +}; + +/// Where the group/timer's source-select and gate fields live. Both are in one word per group, and +/// the bit positions differ per timer, so this is a genuine per-instance lookup rather than a stride. +const TimerClock = struct { + reg: Reg, + src_sel: Field, + clk_en: Field, +}; + +inline fn timerClock(comptime g: Group, comptime t: Timer) TimerClock { + return switch (g) { + .timg0 => switch (t) { + .t0 => .{ + .reg = peri_clk_ctrl20, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T0_CLK_EN_V), + }, + .t1 => .{ + .reg = peri_clk_ctrl20, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP0_T1_CLK_EN_V), + }, + }, + .timg1 => switch (t) { + .t0 => .{ + .reg = peri_clk_ctrl21, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T0_CLK_EN_V), + }, + .t1 => .{ + .reg = peri_clk_ctrl21, + .src_sel = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_SRC_SEL_V), + .clk_en = Field.of(regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_CLK_EN_S, regs.HP_SYS_CLKRST_REG_TIMERGRP1_T1_CLK_EN_V), + }, + }, + }; +} + +/// Select a timer's function clock. Comptime instance because the field pairing really does differ +/// per (group, timer) - four different bit positions in two registers. +pub fn setClockSource(comptime g: Group, comptime t: Timer, src: ClockSource) void { + const c = comptime timerClock(g, t); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + c.reg.modify(.{c.src_sel.is(@intFromEnum(src))}); +} + +/// The timer's function-clock gate, distinct from the group's bus clock in `clkrst`. Defaults to 1 +/// at power-on (hp_sys_clkrst_reg.h: REG_TIMERGRP0_T0_CLK_EN default 1). +pub fn setClockEnabled(comptime g: Group, comptime t: Timer, on: bool) void { + const c = comptime timerClock(g, t); + const guard = clkrst.maskInterrupts(); + defer guard.release(); + c.reg.modify(.{c.clk_en.is(@intFromBool(on))}); +} + +// ------------------------------------------------------------------------ general purpose timer + +pub const Direction = enum { up, down }; + +/// Prescaler on the function clock. 2 is the smallest the hardware accepts and 65536 the largest, +/// encoded as 0 (timer_ll.h:191-199). The divider counter is reset in a second store afterwards, +/// exactly as IDF does it: without that the new divider only takes effect after the old one's +/// current period ends, so the first tick after a change is the wrong length. +pub fn setDivider(g: Group, t: Timer, div: u32) void { + std.debug.assert(div >= 2 and div <= 65536); + const cfg = tReg(a_config, g, t); + cfg.modify(.{divider.is(if (div >= 65536) 0 else div)}); + cfg.modify(.{divcnt_rst.is(1)}); +} + +pub fn setDirection(g: Group, t: Timer, dir: Direction) void { + tReg(a_config, g, t).modify(.{increase.is(@intFromBool(dir == .up))}); +} + +/// Reload the counter from `TxLOADLO`/`TxLOADHI` automatically on every alarm. +pub fn setAutoReload(g: Group, t: Timer, on: bool) void { + tReg(a_config, g, t).modify(.{autoreload.is(@intFromBool(on))}); +} + +pub fn setCounterEnabled(g: Group, t: Timer, on: bool) void { + tReg(a_config, g, t).modify(.{counter_en.is(@intFromBool(on))}); +} + +pub fn setAlarmEnabled(g: Group, t: Timer, on: bool) void { + tReg(a_config, g, t).modify(.{alarm_en.is(@intFromBool(on))}); +} + +/// The 54-bit alarm value. Low word first would be equally correct - the comparator only sees the +/// pair - but IDF writes high then low (timer_ll.h:279-283) and matching its order keeps the write +/// trace comparable. +pub fn setAlarmValue(g: Group, t: Timer, value: u64) void { + tReg(a_alarm_hi, g, t).modify(.{alarm_value_hi.is(@truncate(value >> 32))}); + tReg(a_alarm_lo, g, t).writeRaw(@truncate(value)); +} + +/// The value a reload puts into the counter, whether triggered by `load` or by an auto-reload. +pub fn setLoadValue(g: Group, t: Timer, value: u64) void { + tReg(a_load_hi, g, t).modify(.{load_value_hi.is(@truncate(value >> 32))}); + tReg(a_load_lo, g, t).writeRaw(@truncate(value)); +} + +pub fn getLoadValue(g: Group, t: Timer) u64 { + const hi: u64 = tReg(a_load_hi, g, t).get(load_value_hi); + return (hi << 32) | tReg(a_load_lo, g, t).raw(); +} + +/// Copy the load value into the counter now. `TIMG_TxLOAD_REG` is a whole-word write-to-trigger +/// register: the value written is irrelevant, so this is a bare store rather than a field write. +pub fn load(g: Group, t: Timer) void { + tReg(a_load, g, t).writeRaw(1); +} + +/// The counter, through the capture handshake described at the top of this file. +/// +/// Returns null rather than spinning forever if the peripheral never acknowledges: with the group's +/// bus clock gated off, or the timer's function clock gated off, `UPDATE` never clears, and hanging +/// inside a HAL call with no output is the worst possible way to report that. The bound is the same +/// 10,000 reads `systimer.read` uses. +pub fn read(g: Group, t: Timer) ?u64 { + const upd = tReg(a_update, g, t); + + // Ask for a capture. IDF assigns to the struct bitfield, which is a read-modify-write of a word + // whose only other bits are reserved, so `modify` is both the honest operation and the one that + // produces the same store. + upd.modify(.{update.is(1)}); + + var spins: u32 = 0; + while (upd.get(update) != 0) { + spins += 1; + if (spins > 10_000) return null; + } + + const hi: u64 = tReg(a_hi, g, t).get(count_hi); + return (hi << 32) | tReg(a_lo, g, t).raw(); +} + +// ----------------------------------------------------------------------------------- watchdog + +const a_wdtconfig0 = mmio.addr(regs.TIMG_WDTCONFIG0_REG(0)); +const a_wdtconfig1 = mmio.addr(regs.TIMG_WDTCONFIG1_REG(0)); +const a_wdtconfig2 = mmio.addr(regs.TIMG_WDTCONFIG2_REG(0)); +const a_wdtconfig3 = mmio.addr(regs.TIMG_WDTCONFIG3_REG(0)); +const a_wdtconfig4 = mmio.addr(regs.TIMG_WDTCONFIG4_REG(0)); +const a_wdtconfig5 = mmio.addr(regs.TIMG_WDTCONFIG5_REG(0)); +const a_wdtfeed = mmio.addr(regs.TIMG_WDTFEED_REG(0)); +const a_wdtwprotect = mmio.addr(regs.TIMG_WDTWPROTECT_REG(0)); + +const wdt_en = Field.of(regs.TIMG_WDT_EN_S, regs.TIMG_WDT_EN_V); +const wdt_conf_update_en = Field.of(regs.TIMG_WDT_CONF_UPDATE_EN_S, regs.TIMG_WDT_CONF_UPDATE_EN_V); +const wdt_flashboot_mod_en = Field.of(regs.TIMG_WDT_FLASHBOOT_MOD_EN_S, regs.TIMG_WDT_FLASHBOOT_MOD_EN_V); +const wdt_cpu_reset_length = Field.of(regs.TIMG_WDT_CPU_RESET_LENGTH_S, regs.TIMG_WDT_CPU_RESET_LENGTH_V); +const wdt_sys_reset_length = Field.of(regs.TIMG_WDT_SYS_RESET_LENGTH_S, regs.TIMG_WDT_SYS_RESET_LENGTH_V); +const wdt_clk_prescale = Field.of(regs.TIMG_WDT_CLK_PRESCALE_S, regs.TIMG_WDT_CLK_PRESCALE_V); +const wdt_divcnt_rst = Field.of(regs.TIMG_WDT_DIVCNT_RST_S, regs.TIMG_WDT_DIVCNT_RST_V); + +/// The write-protect key, and also `TIMG_WDT_WKEY`'s reset value: protection is on whenever the +/// register holds anything *else* (timer_group_reg.h, TIMG_WDT_WKEY, default 1356348065). IDF's +/// `mwdt_ll_write_protect_disable` writes this exact constant (mwdt_ll.h:243). +pub const wkey: u32 = 0x50D8_3AA1; + +/// What IDF writes to re-enable protection (mwdt_ll.h:233). Any non-key value would do; using the +/// same one keeps the register comparable against IDF's. +const wkey_locked: u32 = 0; + +// The headers carry no reset-value macro to check `wkey` against - `TIMG_WDT_WKEY_V` is the field +// mask, 0xffffffff - so the constant is copied from the two places that state it: the register +// description's "default: 1356348065" and mwdt_ll.h:243's 0x50D83AA1. The unit test at the end of +// this file pins those two against each other, which is the only check available without a chip. + +/// MWDT stages, each with its own timeout and its own action. Stage 0 fires first; a stage that is +/// not fed escalates to the next. +pub const Stage = enum(u2) { stage0 = 0, stage1 = 1, stage2 = 2, stage3 = 3 }; + +/// What a stage does when it expires (mwdt_ll.h:23-26). +pub const Action = enum(u2) { + off = 0, + interrupt = 1, + reset_cpu = 2, + reset_system = 3, +}; + +/// Length of the reset pulse a `reset_cpu`/`reset_system` stage asserts (mwdt_ll.h:28-35). +pub const ResetLength = enum(u3) { + ns_100 = 0, + ns_200 = 1, + ns_300 = 2, + ns_400 = 3, + ns_500 = 4, + ns_800 = 5, + us_1_6 = 6, + us_3_2 = 7, +}; + +/// A group's MWDT with write protection lifted, and the only way to reach an MWDT operation: +/// +/// const wdt = timg.unlock(.timg1); +/// defer wdt.release(); +/// wdt.setStage(.stage0, ticks, .reset_system); +/// +/// The handle exists because a watchdog register write made while protection is on is silently +/// dropped - no fault, no status bit, just a watchdog that keeps its old timeout - and that is not a +/// mistake worth making twice. +pub const Watchdog = struct { + group: Group, + + /// Re-enable write protection. Not idempotent-with-`unlock` in the composable sense that + /// `clkrst.Guard` is: the hardware has one key register and no nesting count, so an inner + /// `release` really does lock an outer caller out. There is nothing in this HAL that nests. + pub inline fn release(self: Watchdog) void { + gReg(a_wdtwprotect, self.group).writeRaw(wkey_locked); + } + + /// WDTCONFIG0-5 are shadowed; the hardware only takes them at a `CONF_UPDATE_EN` pulse + /// (mwdt_ll.h:121-122). Write-to-trigger, so this is a single deliberate store. + inline fn commit(self: Watchdog) void { + gReg(a_wdtconfig0, self.group).modify(.{wdt_conf_update_en.is(1)}); + } + + inline fn config0(self: Watchdog) Reg { + return gReg(a_wdtconfig0, self.group); + } + + /// The stage's action bits and its timeout live in different registers - the action in + /// WDTCONFIG0, the timeout in WDTCONFIG2+stage - which is why this takes both at once + /// (mwdt_ll.h:98-123). `timeout` is in MWDT clock cycles, i.e. after the prescaler. + pub fn setStage(self: Watchdog, stage: Stage, timeout: u32, action: Action) void { + self.config0().modify(.{stageAction(stage).is(@intFromEnum(action))}); + gReg(stageHoldAddr(stage), self.group).writeRaw(timeout); + self.commit(); + } + + /// Turn one stage off without disturbing its timeout (mwdt_ll.h:131-152). + pub fn disableStage(self: Watchdog, stage: Stage) void { + self.config0().modify(.{stageAction(stage).is(@intFromEnum(Action.off))}); + self.commit(); + } + + pub fn getStageTimeout(self: Watchdog, stage: Stage) u32 { + return gReg(stageHoldAddr(stage), self.group).raw(); + } + + /// Prescaler from the MWDT's source clock (XTAL on this chip - mwdt_ll.h:273-283 asserts it and + /// selects nothing). 1 to 65535; IDF's default is 20000, which gives 500 ticks/us + /// (mwdt_ll.h:20). + pub fn setPrescaler(self: Watchdog, prescaler: u32) void { + std.debug.assert(prescaler >= 1 and prescaler <= 0xffff); + gReg(a_wdtconfig1, self.group).modify(.{wdt_clk_prescale.is(prescaler)}); + self.commit(); + } + + pub fn setCpuResetLength(self: Watchdog, len: ResetLength) void { + self.config0().modify(.{wdt_cpu_reset_length.is(@intFromEnum(len))}); + self.commit(); + } + + pub fn setSysResetLength(self: Watchdog, len: ResetLength) void { + self.config0().modify(.{wdt_sys_reset_length.is(@intFromEnum(len))}); + self.commit(); + } + + /// Flash-boot protection: a second, independent way for this watchdog to run. It ignores + /// `WDT_EN` entirely (mwdt_ll.h:186-188), it defaults to 1, and a group reset re-arms it - so + /// clearing it is part of every sane bring-up, and `clkrst.resetPeripheral` does it. + pub fn setFlashbootEnabled(self: Watchdog, on: bool) void { + self.config0().modify(.{wdt_flashboot_mod_en.is(@intFromBool(on))}); + self.commit(); + } + + /// Start or stop the watchdog. No `CONF_UPDATE_EN` pulse: `mwdt_ll_enable` and `_disable` + /// (mwdt_ll.h:61-77) do not, so neither does this. Disabling does *not* stop flash-boot mode. + pub fn setEnabled(self: Watchdog, on: bool) void { + self.config0().modify(.{wdt_en.is(@intFromBool(on))}); + } + + pub fn isEnabled(self: Watchdog) bool { + return self.config0().get(wdt_en) == 1; + } + + /// Reset the count and the stage. `TIMG_WDTFEED_REG` is a whole-word write-to-trigger register, + /// so the value is irrelevant (mwdt_ll.h:219-222). + pub fn feed(self: Watchdog) void { + gReg(a_wdtfeed, self.group).writeRaw(1); + } + + /// Reset the watchdog's clock divider counter. Write-to-trigger, in WDTCONFIG1 alongside the + /// prescaler. + pub fn resetDividerCount(self: Watchdog) void { + gReg(a_wdtconfig1, self.group).modify(.{wdt_divcnt_rst.is(1)}); + self.commit(); + } +}; + +/// Lift write protection and hand back the only handle that can touch the MWDT. +pub fn unlock(g: Group) Watchdog { + gReg(a_wdtwprotect, g).writeRaw(wkey); + return .{ .group = g }; +} + +/// Feed a watchdog, protection dance included. The one MWDT operation that is worth a shortcut, +/// because it is the one called from a loop. +pub fn feed(g: Group) void { + const wdt = unlock(g); + defer wdt.release(); + wdt.feed(); +} + +/// True if write protection is currently on, i.e. the key register holds something other than the +/// key. Reads the register, so it reports the hardware rather than what this module last wrote. +pub fn isWriteProtected(g: Group) bool { + return gReg(a_wdtwprotect, g).raw() != wkey; +} + +inline fn stageAction(stage: Stage) Field { + // Stage 0 is at the *top* of the word (bits 30:29) and stage 3 at 24:23, i.e. the stages run + // downwards through the register. The four are a uniform 2 bits apart, but in the reverse of + // the obvious direction, so they are looked up rather than computed. + return switch (stage) { + .stage0 => Field.of(regs.TIMG_WDT_STG0_S, regs.TIMG_WDT_STG0_V), + .stage1 => Field.of(regs.TIMG_WDT_STG1_S, regs.TIMG_WDT_STG1_V), + .stage2 => Field.of(regs.TIMG_WDT_STG2_S, regs.TIMG_WDT_STG2_V), + .stage3 => Field.of(regs.TIMG_WDT_STG3_S, regs.TIMG_WDT_STG3_V), + }; +} + +inline fn stageHoldAddr(stage: Stage) u32 { + // WDTCONFIG2 holds stage 0's timeout and WDTCONFIG5 stage 3's; the mapping is off by two and + // there is no macro that says so, so it comes from mwdt_ll.h:100-116. + return switch (stage) { + .stage0 => a_wdtconfig2, + .stage1 => a_wdtconfig3, + .stage2 => a_wdtconfig4, + .stage3 => a_wdtconfig5, + }; +} + +test "the two strides are the documented ones" { + // 0x1000 between groups (timer_group_reg.h:14, REG_TIMG_BASE) and 0x24 between the two timers + // of a group. Both are asserted against the macros at comptime above; this pins the numbers so + // a header change shows up as a failing test with a value in it, not only as a compile error. + try std.testing.expectEqual(@as(u32, 0x1000), group_stride); + try std.testing.expectEqual(@as(u32, 0x24), timer_stride); +} + +test "the write-protect key is the register's reset value" { + try std.testing.expectEqual(@as(u32, 1_356_348_065), wkey); +} diff --git a/src/hal/uart.zig b/src/hal/uart.zig new file mode 100644 index 0000000..7c891a2 --- /dev/null +++ b/src/hal/uart.zig @@ -0,0 +1,622 @@ +//! The HP UART controllers: UART0-4. +//! +//! Out of scope on purpose: UHCI/DMA, RS485, IrDA, hardware and software flow control, the wakeup +//! machinery, and LP_UART (which is a different block behind a different clock tree, not an +//! instance of this one). +//! +//! Three things about this peripheral cost real debugging time, and all three are structural rather +//! than incidental: +//! +//! **Half the configuration registers are shadowed.** The registers whose macro name ends `_SYNC` - +//! UART_CLKDIV_SYNC, UART_CONF0_SYNC, and a dozen more - are not the live configuration. A write +//! lands in a shadow that the core clock domain ignores until UART_REG_UPDATE is set, at which +//! point the hardware copies the shadow across and clears the bit itself. Reads come back from the +//! shadow, so a read-modify-write composes correctly and a read-back proves nothing about what the +//! transmitter is currently using. Every mutator here therefore ends in `update()`, which is +//! exactly what ESP-IDF does: `uart_ll_update` (uart_ll.h:85-89) sets the bit and spins on it, and +//! every `*_sync` writer in that file calls it (set_stop_bits at uart_ll.h:793, set_parity at 828, +//! set_data_bit_num at 1023, set_loop_back at 1439, the FIFO resets at 735 and 750). Omitting it +//! does not fail loudly: the register reads back as asked and the wire keeps the old setting. +//! +//! **Reading offset 0x000 pops the RX FIFO.** `UART_FIFO_REG`'s only field is annotated `RO` in +//! uart_reg.h:18 and that annotation is wrong in the way that matters - the read is the pop. A +//! generic "snapshot the block" loop therefore eats received bytes, which is why the differential +//! harness carries a per-peripheral deny-list of offsets. Writes to the same address push a byte, +//! and must be full 32-bit stores: a byte store on this bus is a read-modify-write, so it would pop +//! a byte in order to push one (uart_ll.h:716-724 says so and is the reason `pushByte` uses +//! `writeRaw`). +//! +//! **UART0 is the console.** Resetting it clears UART_CLKDIV, the console turns to garbage +//! mid-sentence and the board dies on a watchdog reset with nothing readable to explain it. That was +//! measured on this board. Nothing here resets UART0 implicitly, `reset()` refuses instance 0, and +//! the differential suite uses UART1. +//! +//! The clock path is two dividers in series and they live in different blocks: HP_SYS_CLKRST holds +//! the integer pre-divider (`REG_UARTn_SCLK_DIV_NUM`) and the source select, the UART itself holds +//! the 12.4 fixed-point divider. `setBaudrate` drives both, because neither alone spans the range. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// UART0-4. LP_UART (ESP-IDF's port 5) is a separate peripheral and not modelled here. +pub const count = 5; + +/// SOC_UART_FIFO_LEN, soc_caps.h:655. Both directions; the TX count register reports how many bytes +/// are queued, so free space is this minus that. +pub const fifo_len = 128; + +// ------------------------------------------------------------------------------ register blocks +// One 0x1000-byte block per instance (soc.h:20, `REG_UART_BASE(i) = DR_REG_UART_BASE + i*0x1000`). +// Every register is reached through a RegArray so the stride is checked against the headers rather +// than assumed, and a wrong instance index is a bounds assert rather than a write into UART2. + +fn regArray(comptime offset: u32) type { + return mmio.RegArray( + regs.DR_REG_UART0_BASE + offset, + regs.DR_REG_UART0_BASE + 0x1000 + offset, + count, + ); +} + +const fifo = regArray(0x00); +const clkdiv_sync = regArray(0x14); +const status = regArray(0x1c); +const conf0_sync = regArray(0x20); +const clk_conf = regArray(0x88); +const reg_update = regArray(0x98); + +// CLKDIV_SYNC: a 12.4 fixed-point divider, with the fraction not adjacent to the integer part. +const clkdiv = Field.of(regs.UART_CLKDIV_S, regs.UART_CLKDIV_V); +const clkdiv_frag = Field.of(regs.UART_CLKDIV_FRAG_S, regs.UART_CLKDIV_FRAG_V); + +// CONF0_SYNC: the data format, the FIFO resets and the loopback switch all share this word, which is +// why every one of them is a read-modify-write and not a `write`. +const parity = Field.of(regs.UART_PARITY_S, regs.UART_PARITY_V); +const parity_en = Field.of(regs.UART_PARITY_EN_S, regs.UART_PARITY_EN_V); +const bit_num = Field.of(regs.UART_BIT_NUM_S, regs.UART_BIT_NUM_V); +const stop_bit_num = Field.of(regs.UART_STOP_BIT_NUM_S, regs.UART_STOP_BIT_NUM_V); +const loopback = Field.of(regs.UART_LOOPBACK_S, regs.UART_LOOPBACK_V); +const rxfifo_rst = Field.of(regs.UART_RXFIFO_RST_S, regs.UART_RXFIFO_RST_V); +const txfifo_rst = Field.of(regs.UART_TXFIFO_RST_S, regs.UART_TXFIFO_RST_V); + +// STATUS: live counters, so read-only and never worth comparing between two runs. +const rxfifo_cnt = Field.of(regs.UART_RXFIFO_CNT_S, regs.UART_RXFIFO_CNT_V); +const txfifo_cnt = Field.of(regs.UART_TXFIFO_CNT_S, regs.UART_TXFIFO_CNT_V); + +const tx_sclk_en = Field.of(regs.UART_TX_SCLK_EN_S, regs.UART_TX_SCLK_EN_V); +const rx_sclk_en = Field.of(regs.UART_RX_SCLK_EN_S, regs.UART_RX_SCLK_EN_V); + +/// UART_REG_UPDATE, the commit bit for the whole `_SYNC` family. `R/W/SC`: the hardware clears it +/// when the copy is done. +const reg_update_bit = Field.of(regs.UART_REG_UPDATE_S, regs.UART_REG_UPDATE_V); + +// -------------------------------------------------------------------------------- clock control +// The source select and the integer pre-divider are one register apart, and not in the register the +// names suggest: for UARTn the select is in PERI_CLK_CTRL(110+n) and the pre-divider is in +// PERI_CLK_CTRL(111+n). That is not a typo in this file - uart_ll.h:463-475 writes +// `peri_clk_ctrl110.reg_uart0_clk_src_sel` while uart_ll.h:558-568 writes +// `peri_clk_ctrl111.reg_uart0_sclk_div_num`, so ctrl111 holds UART0's divider *and* UART1's select. + +const peri_clk_ctrl = mmio.RegArray( + regs.HP_SYS_CLKRST_PERI_CLK_CTRL110_REG, + regs.HP_SYS_CLKRST_PERI_CLK_CTRL111_REG, + 6, // ctrl110..ctrl115: five selects and five dividers, overlapping by one +); + +// All five instances place these fields at the same shifts in their respective registers +// (hp_sys_clkrst_reg.h: every REG_UARTn_CLK_SRC_SEL_S is 24, every REG_UARTn_SCLK_DIV_NUM_S is 0, +// every REG_UARTn_CLK_EN_S is 26), so one macro triple each describes all of them. +const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_V); +const sclk_div_num = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_V); +const sclk_en = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_V); + +/// The three clock sources an HP UART can take, with the encoding from uart_ll.h:447-461. +pub const ClockSource = enum(u2) { + /// The 40 MHz crystal. The only source whose frequency is exact, which is why it is the default + /// for anything that has to interoperate. + xtal = 0, + /// RC_FAST, the always-on oscillator. Nominally 20 MHz and uncalibrated - it varies with + /// temperature and part, so a baud rate derived from `nominalHz` here is approximate. + rtc = 1, + /// A fixed 80 MHz tap off the system PLL. Needed for the high rates: the 12-bit integer divider + /// runs out below about 5 kBd from XTAL. + pll_f80m = 2, + + /// The nominal frequency to hand `setBaudrate`. Nominal is exact for `xtal` and `pll_f80m` and a + /// datasheet typical for `rtc`; the real clock tree can be reconfigured, so a caller that has + /// changed it must pass its own number instead. + pub fn nominalHz(self: ClockSource) u32 { + return switch (self) { + .xtal => 40_000_000, + .rtc => 20_000_000, + .pll_f80m => 80_000_000, + }; + } +}; + +pub const WordLength = enum(u2) { + // uart_types.h:57-60. The encoding is (bits - 5), which is why it starts at zero. + bits5 = 0, + bits6 = 1, + bits7 = 2, + bits8 = 3, +}; + +pub const StopBits = enum(u2) { + // uart_types.h:68-70. There is no encoding for zero stop bits, so the enum starts at 1 and 0 is + // reserved by the hardware. + one = 1, + one_and_half = 2, + two = 3, +}; + +pub const Parity = enum(u2) { + // uart_types.h:78-80: bit 1 is "parity enabled", bit 0 is odd/even. `disable` is 0, so the + // odd/even bit is not part of it - see `setParity` for why that matters. + disable = 0, + even = 2, + odd = 3, +}; + +/// One UART instance. A value type holding nothing but the index, so it costs nothing at runtime and +/// every register access folds to a constant address when the index is known. +pub const Uart = struct { + num: u8, + + pub fn init(num: u8) Uart { + std.debug.assert(num < count); + return .{ .num = num }; + } + + // ----------------------------------------------------------------------------- the commit bit + + /// Copy the `_SYNC` shadow registers into the core clock domain and wait for the hardware to + /// acknowledge by clearing the bit (uart_ll.h:85-89). + /// + /// Bounded, where ESP-IDF's `while (hw->reg_update.reg_update);` is not: a UART whose core clock + /// is gated off never clears the bit, and on a board with no debugger an infinite spin is + /// indistinguishable from a crash. 4096 spins is several thousand times the observed cost of a + /// commit, which takes a handful of core-clock cycles. Returns false rather than panicking so a + /// caller can report the peripheral instead of losing the console. + pub fn update(self: Uart) bool { + const r = reg_update.at(self.num); + r.modify(.{reg_update_bit.is(1)}); + return r.waitFor(reg_update_bit, 0, 4096); + } + + // ---------------------------------------------------------------------------- clocks and reset + + /// Reset the block. Refuses UART0. + /// + /// UART0 carries this board's console. A reset clears UART_CLKDIV to its power-on 694, the + /// console's output becomes garbage part-way through whatever it was printing, and the board + /// takes a watchdog reset a moment later - measured, not theorised. There is no "and then put + /// the divider back" version of this that is safe, because the damage is done between the two + /// stores. + pub fn reset(self: Uart) void { + std.debug.assert(self.num != 0); + switch (self.num) { + 1 => clkrst.resetPeripheral(.uart1), + 2 => clkrst.resetPeripheral(.uart2), + 3 => clkrst.resetPeripheral(.uart3), + 4 => clkrst.resetPeripheral(.uart4), + else => unreachable, + } + } + + /// The core (baud-generating) clock, as distinct from the APB bus clock that + /// `clkrst.setClockEnabled` handles. Both are needed: the bus clock makes the registers + /// answer, this one makes the shift registers move - and `update()` is one of the things that + /// stops working without it. + /// + /// Two gates in two blocks, per uart_ll.h:379-397: HP_SYS_CLKRST's per-instance `CLK_EN`, which + /// sits in the *select* register PERI_CLK_CTRL(110+n) and not the divider one next to it, and + /// the UART's own TX and RX enables in UART_CLK_CONF. Interrupts are masked over the first + /// because PERI_CLK_CTRL is shared with unrelated peripherals. + pub fn setCoreClockEnabled(self: Uart, on: bool) void { + const v: u32 = @intFromBool(on); + { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.selectReg().modify(.{sclk_en.is(v)}); + } + clk_conf.at(self.num).modify(.{ tx_sclk_en.is(v), rx_sclk_en.is(v) }); + } + + /// Select the clock the baud generator divides down. Read-modify-write of a register shared with + /// other peripherals, so interrupts are masked (uart_ll.h:477-481 makes the equivalent point by + /// refusing to compile outside `PERIPH_RCC_ATOMIC`). + pub fn setClockSource(self: Uart, src: ClockSource) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.selectReg().modify(.{clk_src_sel.is(@intFromEnum(src))}); + } + + pub fn clockSource(self: Uart) ClockSource { + // Encoding 3 is not defined; IDF's getter (uart_ll.h:509-524) maps `default` to RTC, so + // reporting the same thing keeps a round-trip through both implementations consistent. + return switch (self.selectReg().get(clk_src_sel)) { + 0 => .xtal, + 2 => .pll_f80m, + else => .rtc, + }; + } + + /// PERI_CLK_CTRL(110+n): where this instance's source select and core clock gate live. + inline fn selectReg(self: Uart) Reg { + return peri_clk_ctrl.at(self.num); + } + + /// PERI_CLK_CTRL(111+n): where this instance's integer pre-divider lives. One register above + /// the select, which is the trap this pair of accessors exists to contain. + inline fn dividerReg(self: Uart) Reg { + return peri_clk_ctrl.at(self.num + 1); + } + + // ------------------------------------------------------------------------------------- baud + + /// The two dividers a baud rate decomposes into, computed exactly as + /// `_uart_ll_set_baudrate` (uart_ll.h:532-588) does. + pub const Divider = struct { + /// HP_SYS_CLKRST's integer pre-divider, 1-256. Stored as `sclk - 1` in an 8-bit field. + sclk: u32, + /// The UART's own divider, integer part, 12 bits. + int: u32, + /// The UART's own divider, sixteenths. + frag: u32, + }; + + /// Decompose a baud rate, or fail if the hardware cannot express it. + /// + /// The arithmetic, line by line against uart_ll.h: + /// + /// 541 max_div = UART_CLKDIV_V = 0xfff - the UART divider's integer part is 12 bits + /// 542 sclk = ceil(sclk_freq / (max_div * baud)) the smallest pre-divide that brings + /// the remaining ratio inside 12 bits + /// 545 reject sclk == 0 or sclk > 256 256 = SCLK_DIV_NUM_V + 1 + /// 549 clk_div = (sclk_freq << 4) / (baud * sclk) the ratio in sixteenths + /// 551 int = clk_div >> 4 + /// 552 frag = clk_div & 0xf + /// 555+ the field written is sclk - 1 + /// + /// The `<< 4` is IDF's fixed-point scale, not a fudge: CLKDIV_FRAG is a count of sixteenths of a + /// source-clock period added to every bit time, so `clk_div` is the exact ratio rounded down to + /// 1/16 of a tick. Two deliberate departures from the C, neither of which changes a result: + /// + /// * The `ceil` denominator is 64-bit here as it is there (uart_ll.h:542 casts `max_div` to + /// `uint64_t`), and `sclk_freq << 4` is *also* computed in 64 bits. In C that shift is + /// `uint32_t` and overflows above 268.4 MHz; no P4 UART source is anywhere near that (the + /// fastest is PLL_F80M at 80 MHz), so the two agree on every reachable input while this one + /// has no undefined case. + /// * `baud == 0` returns null rather than false-with-registers-untouched; same outcome, but the + /// caller cannot ignore it by accident. + pub fn divider(baud: u32, sclk_freq: u32) ?Divider { + if (baud == 0) return null; + const max_div: u64 = clkdiv.max(); // UART_CLKDIV_V + const denom = max_div * baud; + const sclk: u64 = (@as(u64, sclk_freq) + denom - 1) / denom; + if (sclk == 0 or sclk > @as(u64, sclk_div_num.max()) + 1) return null; + const clk_div: u64 = (@as(u64, sclk_freq) << 4) / (@as(u64, baud) * sclk); + return .{ + .sclk = @intCast(sclk), + .int = @intCast(clk_div >> 4), + .frag = @intCast(clk_div & 0xf), + }; + } + + /// Program a baud rate. Returns false, having touched nothing, if it is unreachable from this + /// source frequency. + /// + /// Store order follows uart_ll.h:550-576 exactly - integer part, fraction, pre-divider, commit - + /// because the intermediate states are visible to the transmitter of a UART that is already + /// running, and because a write-trace comparison against IDF would otherwise differ on ordering + /// while agreeing on the final registers. The two CLKDIV_SYNC stores are separate for the same + /// reason: IDF's two bitfield assignments are two read-modify-writes of that word. + pub fn setBaudrate(self: Uart, baud: u32, sclk_freq: u32) bool { + const d = divider(baud, sclk_freq) orelse return false; + const div = clkdiv_sync.at(self.num); + div.modify(.{clkdiv.is(d.int)}); + div.modify(.{clkdiv_frag.is(d.frag)}); + { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.dividerReg().modify(.{sclk_div_num.is(d.sclk - 1)}); + } + _ = self.update(); + return true; + } + + /// The baud rate the registers currently describe, by inverting the above + /// (uart_ll.h:590-615). Integer division both ways, so this is not exactly the value passed to + /// `setBaudrate` - it is what the hardware will actually produce, which is the more useful + /// number. + pub fn baudrate(self: Uart, sclk_freq: u32) u32 { + const div = clkdiv_sync.at(self.num).raw(); + const int = (div >> clkdiv.shift) & clkdiv.unshiftedMask(); + const frag = (div >> clkdiv_frag.shift) & clkdiv_frag.unshiftedMask(); + const sclk = self.dividerReg().get(sclk_div_num) + 1; + const ticks = ((@as(u64, int) << 4) | frag) * sclk; + if (ticks == 0) return 0; + return @intCast((@as(u64, sclk_freq) << 4) / ticks); + } + + // ------------------------------------------------------------------------------ data format + + /// uart_ll.h:1020-1024. + pub fn setWordLength(self: Uart, w: WordLength) void { + conf0_sync.at(self.num).modify(.{bit_num.is(@intFromEnum(w))}); + _ = self.update(); + } + + /// uart_ll.h:790-794. + pub fn setStopBits(self: Uart, s: StopBits) void { + conf0_sync.at(self.num).modify(.{stop_bit_num.is(@intFromEnum(s))}); + _ = self.update(); + } + + /// uart_ll.h:817-832. + /// + /// Note what IDF does *not* do: disabling parity leaves UART_PARITY - the odd/even select bit - + /// at whatever it was, because the value 0 for "disabled" carries no odd/even information and + /// writing bit 0 of it would be writing a zero the caller never asked for. So `.disable` clears + /// `parity_en` only. Reproduced here because otherwise a differential run diverges by one bit + /// after any sequence that sets odd parity and then disables it. + pub fn setParity(self: Uart, p: Parity) void { + const c = conf0_sync.at(self.num); + const v = @intFromEnum(p); + if (p != .disable) c.modify(.{parity.is(v & 1)}); + c.modify(.{parity_en.is((v >> 1) & 1)}); + _ = self.update(); + } + + /// All three format fields, in IDF's order. Three commits rather than one, matching what + /// calling IDF's three setters does: the format of a UART mid-transmission is not atomic on + /// this hardware either way, and diverging here would be a difference with no benefit. + pub fn setFormat(self: Uart, w: WordLength, p: Parity, s: StopBits) void { + self.setWordLength(w); + self.setParity(p); + self.setStopBits(s); + } + + pub fn wordLength(self: Uart) WordLength { + return @enumFromInt(conf0_sync.at(self.num).get(bit_num)); + } + + pub fn stopBits(self: Uart) StopBits { + // Encoding 0 is not a legal stop-bit count. The hardware's reset value is 1, and nothing + // here can write 0, so an out-of-range read means the block is unclocked or was reset + // under us - reported as `one` rather than an illegal enum value, which would be UB. + return switch (conf0_sync.at(self.num).get(stop_bit_num)) { + 2 => .one_and_half, + 3 => .two, + else => .one, + }; + } + + /// uart_ll.h:834-841: parity is only meaningful when enabled, so the odd/even bit is not + /// reported unless it is. + pub fn parityMode(self: Uart) Parity { + const c = conf0_sync.at(self.num).raw(); + if ((c >> parity_en.shift) & 1 == 0) return .disable; + return if ((c >> parity.shift) & 1 == 1) .odd else .even; + } + + // ------------------------------------------------------------------------------------- FIFO + + /// Bytes waiting in the RX FIFO (uart_ll.h:763-766). + pub fn rxCount(self: Uart) u32 { + return status.at(self.num).get(rxfifo_cnt); + } + + /// Bytes queued in the TX FIFO. + pub fn txCount(self: Uart) u32 { + return status.at(self.num).get(txfifo_cnt); + } + + /// Free space in the TX FIFO (uart_ll.h:775-780: the total, minus what is queued). + pub fn txFree(self: Uart) u32 { + return fifo_len - self.txCount(); + } + + /// Push one byte. A full 32-bit store, because a narrower one becomes a read-modify-write on + /// this bus and the read would pop a received byte (uart_ll.h:716-724). + pub inline fn pushByte(self: Uart, byte: u8) void { + fifo.at(self.num).writeRaw(byte); + } + + /// Pop one byte. The read *is* the pop - see this file's header on why offset 0x000 is on the + /// differential harness's no-read list. + pub inline fn popByte(self: Uart) u8 { + return @truncate(fifo.at(self.num).raw()); + } + + /// Discard everything received. Assert, commit, deassert, commit: `rxfifo_rst` lives in a + /// shadow register, so without the commits the hardware never sees either edge + /// (uart_ll.h:733-739). + pub fn resetRxFifo(self: Uart) void { + const c = conf0_sync.at(self.num); + c.modify(.{rxfifo_rst.is(1)}); + _ = self.update(); + c.modify(.{rxfifo_rst.is(0)}); + _ = self.update(); + } + + /// uart_ll.h:748-754. Same shape, and the same reason for it. + pub fn resetTxFifo(self: Uart) void { + const c = conf0_sync.at(self.num); + c.modify(.{txfifo_rst.is(1)}); + _ = self.update(); + c.modify(.{txfifo_rst.is(0)}); + _ = self.update(); + } + + // --------------------------------------------------------------------------------- loopback + + /// Tie TX back to RX inside the block (uart_ll.h:1437-1441). The pads are not involved, which + /// makes it the only way to exercise a UART end to end with nothing wired to the board - it is + /// how the FIFO and format paths can be tested at all here. + pub fn setLoopback(self: Uart, on: bool) void { + conf0_sync.at(self.num).modify(.{loopback.is(@intFromBool(on))}); + _ = self.update(); + } + + pub fn loopbackEnabled(self: Uart) bool { + return conf0_sync.at(self.num).get(loopback) == 1; + } + + // ------------------------------------------------------------------------------ pin routing + + /// This instance's TX signal index in the GPIO matrix. The names in IDF's map are + /// `UARTn_TXD_PAD_OUT_IDX` (gpio_sig_map.h:28-52) and they are consecutive in steps of three, + /// but the step is not relied on: each is named. + pub fn txSignal(self: Uart) u32 { + return switch (self.num) { + 0 => regs.UART0_TXD_PAD_OUT_IDX, + 1 => regs.UART1_TXD_PAD_OUT_IDX, + 2 => regs.UART2_TXD_PAD_OUT_IDX, + 3 => regs.UART3_TXD_PAD_OUT_IDX, + 4 => regs.UART4_TXD_PAD_OUT_IDX, + else => unreachable, + }; + } + + /// This instance's RX signal index. Numerically equal to the TX one - the matrix's input and + /// output signal spaces are separate namespaces that happen to share indices for a duplex + /// peripheral - which is exactly why routing RX with `matrixOut` silently does nothing useful. + pub fn rxSignal(self: Uart) u32 { + return switch (self.num) { + 0 => regs.UART0_RXD_PAD_IN_IDX, + 1 => regs.UART1_RXD_PAD_IN_IDX, + 2 => regs.UART2_RXD_PAD_IN_IDX, + 3 => regs.UART3_RXD_PAD_IN_IDX, + 4 => regs.UART4_RXD_PAD_IN_IDX, + else => unreachable, + }; + } + + /// Route TX to a pad through the GPIO matrix. + pub fn routeTx(self: Uart, pin: u8) void { + gpio.matrixOut(pin, self.txSignal()); + } + + /// Route a pad to RX through the GPIO matrix, and enable that pad's input buffer - without + /// which the routed signal reads as a constant and the UART receives nothing, which is the + /// single most common way this goes wrong. + pub fn routeRx(self: Uart, pin: u8) void { + gpio.setInputEnable(pin, true); + gpio.matrixIn(pin, self.rxSignal()); + } + + // --------------------------------------------------------------------------------- transfers + + /// Send every byte, blocking until each fits. Bounded only by the FIFO draining, which always + /// progresses while the core clock is on - so unlike a blocking *read* this cannot wait on an + /// event that may never happen. + pub fn write(self: Uart, bytes: []const u8) void { + for (bytes) |b| { + while (self.txFree() == 0) {} + self.pushByte(b); + } + } + + /// Drain up to `buf.len` received bytes and report how many there were. Does not block. + /// + /// Deliberately not blocking: nothing on the other end of a UART is obliged to send, so a + /// blocking read is an unbounded wait, and there is no timer in this HAL's dependency set to + /// bound it with. A caller that wants to wait writes the loop, and owns the decision about what + /// to do when the bytes never come. + pub fn read(self: Uart, buf: []u8) usize { + var n: usize = 0; + const available = self.rxCount(); + while (n < buf.len and n < available) : (n += 1) buf[n] = self.popByte(); + return n; + } + + /// Whether the transmitter has finished: nothing queued in the FIFO. + /// + /// Not the same as "the last bit is on the wire" - the shift register still holds up to one + /// character after the FIFO empties. UART_FSM_STATUS reports that, and this HAL does not model + /// it, so a caller about to cut the clock or reconfigure the format must allow for one more + /// character time. + pub fn txIdle(self: Uart) bool { + return self.txCount() == 0; + } +}; + +// ------------------------------------------------------------------------------------ host tests +// The divider arithmetic is the only part of this file that can be checked without the chip, and it +// is the part most worth checking: every value below is IDF's formula evaluated by hand, so a +// transcription error in `divider` fails here rather than as a garbled console. + +test "40 MHz XTAL, 115200 Bd: one source tick, 12.4 divider does the work" { + // 40e6/(4095*115200) = 0.085 -> ceil = 1. clk_div = (40e6<<4)/115200 = 5555 (5555.55 floored). + // 5555 = 347*16 + 3. + const d = Uart.divider(115200, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 1), d.sclk); + try std.testing.expectEqual(@as(u32, 347), d.int); + try std.testing.expectEqual(@as(u32, 3), d.frag); + // 347 + 3/16 = 347.1875 ticks per bit -> 115,213 Bd in real arithmetic, and 115,211 as the + // hardware's own truncating inverse reports it (see the round-trip test): 0.01% fast either way. +} + +test "80 MHz PLL, 115200 Bd: the fraction differs from the XTAL case, which is the point of it" { + // 80e6/(4095*115200) = 0.17 -> 1. clk_div = (80e6<<4)/115200 = 11111 = 694*16 + 7. + const d = Uart.divider(115200, 80_000_000).?; + try std.testing.expectEqual(@as(u32, 1), d.sclk); + try std.testing.expectEqual(@as(u32, 694), d.int); + try std.testing.expectEqual(@as(u32, 7), d.frag); +} + +test "a rate low enough to need the pre-divider" { + // 300 Bd from 40 MHz: 40e6/300 = 133,333 ticks per bit, far past 12 bits. + // ceil(40e6/(4095*300)) = ceil(32.6) = 33. clk_div = (40e6<<4)/(300*33) = 64,646 = 4040*16 + 6. + const d = Uart.divider(300, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 33), d.sclk); + try std.testing.expectEqual(@as(u32, 4040), d.int); + try std.testing.expectEqual(@as(u32, 6), d.frag); + try std.testing.expect(d.int <= 0xfff); + try std.testing.expect(d.sclk <= 256); +} + +test "unreachable rates are rejected rather than rounded" { + // Zero is IDF's explicit early return (uart_ll.h:538). + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(0, 40_000_000)); + // 10 Bd from 40 MHz needs a pre-divide of ceil(40e6/40950) = 977, past the 8-bit field's 256. + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(10, 40_000_000)); +} + +test "the sclk == 0 rejection is unreachable except from a zero source frequency" { + // Worth pinning down, because the obvious reading of uart_ll.h:545 is wrong. `sclk` is a + // *ceiling*, so for any non-zero source frequency it is at least 1 - asking for 4 MBd from a + // 1 kHz clock does NOT fail here, it yields sclk = 1 and a divider of zero, and IDF programs + // that just as happily. The only input that trips the branch is sclk_freq == 0. + const absurd = Uart.divider(4_000_000, 1000).?; + try std.testing.expectEqual(@as(u32, 1), absurd.sclk); + try std.testing.expectEqual(@as(u32, 0), absurd.int); + try std.testing.expectEqual(@as(u32, 0), absurd.frag); + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(115200, 0)); +} + +test "the pre-divider field stores sclk - 1, so the reachable rates stop at 256 ticks" { + // The boundary IDF checks at uart_ll.h:545: sclk may be 256 because the field holds sclk-1. + // From 40 MHz the last rate inside it is 39 Bd, at a pre-divide of 251; 38 Bd needs 257. + const ok = Uart.divider(39, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 251), ok.sclk); + try std.testing.expectEqual(@as(u32, 4086), ok.int); + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(38, 40_000_000)); +} + +test "the divider round-trips through the baud rate the hardware will really produce" { + // What `baudrate()` computes, without a chip: the inverse of the same arithmetic. + const d = Uart.divider(115200, 40_000_000).?; + const ticks = ((@as(u64, d.int) << 4) | d.frag) * d.sclk; + const actual: u32 = @intCast((@as(u64, 40_000_000) << 4) / ticks); + // 347 + 3/16 = 347.1875 ticks per bit, and 40e6*16/5555 truncates to 115,211 Bd: 0.01% fast. + try std.testing.expectEqual(@as(u32, 115_211), actual); +} |
