summaryrefslogtreecommitdiff
path: root/src/hal/sdmmc.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-25 12:40:53 -0300
committerGabriel Schneider <[email protected]>2026-08-25 12:46:51 -0300
commitf5f8068fac59b4f16046c2022c2fc7c7e447ef4c (patch)
tree2731a3ed4e51cae09e184e25778eded5fc37d1f5 /src/hal/sdmmc.zig
downloadesp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.tar.gz
esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.zip
zig-p4: pure-Zig ESP32-P4 toolchain
build.zig generates the linker script and drives Zig's own LLD; tools/image.zig turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker. src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers; src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip; src/oracle/ diffs this HAL against ESP-IDF's on the die.
Diffstat (limited to 'src/hal/sdmmc.zig')
-rw-r--r--src/hal/sdmmc.zig2002
1 files changed, 2002 insertions, 0 deletions
diff --git a/src/hal/sdmmc.zig b/src/hal/sdmmc.zig
new file mode 100644
index 0000000..0bebeb0
--- /dev/null
+++ b/src/hal/sdmmc.zig
@@ -0,0 +1,2002 @@
+//! The SDMMC host controller, driven as an **SDIO host**.
+//!
+//! There is no SD card on this board. Slot 1 of the P4's SDMMC controller goes to an ESP32-C6
+//! running ESP-Hosted coprocessor firmware, which presents itself as a 4-bit SDIO device: CLK 18,
+//! CMD 19, D0-D3 = 14/15/16/17. So this file implements CMD0/CMD5/CMD3/CMD7 and then CMD52/CMD53,
+//! and nothing above them. SD memory cards, SPI mode, CSD/CID decoding and block devices are
+//! deliberately absent - they are a different problem that happens to share a peripheral.
+//!
+//! The controller is a Synopsys DesignWare mobile-storage host. Three of its properties decide the
+//! shape of everything below.
+//!
+//! **The card clock is not the register clock.** CLKDIV, CLKSRC and CLKENA are written on the bus
+//! side and do not reach the card-interface unit until a *clock update command* is issued: a write
+//! to the CMD register with `update_clk_reg` and `start_command` set, which sends nothing to the
+//! card (`sdmmc_reg.h:440-456`, and ESP-IDF's `sd_host_slot_clock_update_command`,
+//! `sd_host_sdmmc.c:896-912`). A driver that programmes a divider and moves on has changed
+//! nothing. Three such commands are needed to change frequency safely - clock off, reprogramme,
+//! clock on - and that is what `setBusClock` does.
+//!
+//! **The command register is a single word, and `start_command` is bit 31 of it.** Every attribute
+//! of a command - index, whether a response is expected, whether its CRC is checked, whether data
+//! follows and in which direction - is a field of the same word, and writing that word with bit 31
+//! set launches the command. So the interesting part of "send CMD52" is an encoding, not a
+//! sequence, and `commandWord` is a pure function of the request. It is host-tested, and the
+//! oracle compares the words it produces against words built through ESP-IDF's own
+//! `sdmmc_hw_cmd_t` bitfields.
+//!
+//! **Data moves by internal DMA over descriptors in memory, and the P4 caches that memory.**
+//! `soc_caps.h:185` sets SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE, so L2MEM - where every static in this
+//! image lives - is reached by the CPU through the L1 data cache while the IDMAC reaches it
+//! directly. See the "Cache" section below for the resolution; it is the one place in this file
+//! where the right answer is not visible in any register header.
+//!
+//! Nothing here has been run on hardware by the author of this file. What is claimed is that the
+//! register arithmetic and the command encodings match ESP-IDF's at the cited lines, that
+//! `src/oracle/sdmmc_cases.zig` compares the two on the die, and that the configuration `init`
+//! leaves behind reproduces a dump taken from a working ESP-IDF image on this board.
+
+const std = @import("std");
+const regs = @import("regs");
+const mmio = @import("mmio");
+const gpio = @import("gpio.zig");
+const clkrst = @import("clkrst.zig");
+const intr = @import("intr.zig");
+
+const Reg = mmio.Reg;
+const Field = mmio.Field;
+
+pub const Error = error{ Timeout, CrcError, ResponseError, NotSupported, Busy };
+
+// ------------------------------------------------------------------------------- registers
+//
+// One instance, at DR_REG_SDHOST_BASE = DR_REG_SDMMC_BASE = 0x50083000 (`reg_base.h:44`, `:204`,
+// and `esp32p4.peripherals.ld:41` agrees). The macros are spelled SDHOST_*, the peripheral is
+// spelled SDMMC, and both names are ESP-IDF's.
+
+const ctrl = Reg.at(regs.SDHOST_CTRL_REG);
+const clkdiv = Reg.at(regs.SDHOST_CLKDIV_REG);
+const clksrc = Reg.at(regs.SDHOST_CLKSRC_REG);
+const clkena = Reg.at(regs.SDHOST_CLKENA_REG);
+const tmout = Reg.at(regs.SDHOST_TMOUT_REG);
+const ctype = Reg.at(regs.SDHOST_CTYPE_REG);
+const blksiz = Reg.at(regs.SDHOST_BLKSIZ_REG);
+const bytcnt = Reg.at(regs.SDHOST_BYTCNT_REG);
+const intmask = Reg.at(regs.SDHOST_INTMASK_REG);
+const cmdarg = Reg.at(regs.SDHOST_CMDARG_REG);
+const cmd = Reg.at(regs.SDHOST_CMD_REG);
+const resp0 = Reg.at(regs.SDHOST_RESP0_REG);
+const rintsts = Reg.at(regs.SDHOST_RINTSTS_REG);
+/// The *masked* status: RINTSTS gated by INTMASK, and the only word the controller's interrupt
+/// output looks at. ESP-IDF's `sdmmc_ll_get_intr_status` reads this one and not RINTSTS
+/// (`sdmmc_ll.h:841-844`), which is exactly why INTMASK decides what reaches the CLIC while
+/// RINTSTS stays readable for the polling path.
+const mintsts = Reg.at(regs.SDHOST_MINTSTS_REG);
+const status = Reg.at(regs.SDHOST_STATUS_REG);
+const fifoth = Reg.at(regs.SDHOST_FIFOTH_REG);
+const bmod = Reg.at(regs.SDHOST_BMOD_REG);
+const pldmnd = Reg.at(regs.SDHOST_PLDMND_REG);
+const dbaddr = Reg.at(regs.SDHOST_DBADDR_REG);
+const idsts = Reg.at(regs.SDHOST_IDSTS_REG);
+const idinten = Reg.at(regs.SDHOST_IDINTEN_REG);
+
+// CTRL fields. Two of them - `dma_enable` at bit 5 and `use_internal_dma` at bit 25 - have no
+// `_S`/`_V` macro pair in `sdmmc_reg.h` at all: that header documents CTRL as bits 0,1,2,4,6..11
+// and stops. They are real, they are in the measured working dump (`ctrl=0x02000030`), and
+// `sdmmc_struct.h:76` and `:135` name them at exactly those positions. This is the same situation
+// as the IO MUX pull bits in `hal/gpio.zig`, and the same remedy: `Field.bit` with the struct
+// header cited, because the struct header is ESP-IDF's definition of the layout even where the
+// macro header is incomplete.
+const controller_reset = Field.of(regs.SDHOST_CONTROLLER_RESET_S, regs.SDHOST_CONTROLLER_RESET_V);
+const fifo_reset = Field.of(regs.SDHOST_FIFO_RESET_S, regs.SDHOST_FIFO_RESET_V);
+const dma_reset = Field.of(regs.SDHOST_DMA_RESET_S, regs.SDHOST_DMA_RESET_V);
+const int_enable = Field.of(regs.SDHOST_INT_ENABLE_S, regs.SDHOST_INT_ENABLE_V);
+/// `sdmmc_struct.h:76` - `uint32_t dma_enable:1;` immediately after `int_enable:1` at bit 4.
+const dma_enable = Field.bit(5);
+/// `sdmmc_struct.h:135` - after `reserved2:4`, `card_voltage_a:4`, `card_voltage_b:4` and
+/// `enable_od_pullup:1`, i.e. bit 25. `sdmmc_ll_enable_dma` (`sdmmc_ll.h:812-818`) is the only
+/// writer, and the working dump's `ctrl=0x02000030` has exactly this bit plus 4 and 5.
+const use_internal_dma = Field.bit(25);
+
+const clk_divider0 = Field.of(regs.SDHOST_CLK_DIVIDER0_S, regs.SDHOST_CLK_DIVIDER0_V);
+const clk_divider1 = Field.of(regs.SDHOST_CLK_DIVIDER1_S, regs.SDHOST_CLK_DIVIDER1_V);
+// CLKSRC is documented as one 4-bit field, two bits per card ("bit[1:0] are assigned for card 0,
+// bit[3:2] are assigned for card 1", `sdmmc_reg.h:166-179`). `sdmmc_struct.h:191-192` splits it
+// into `card0:2` and `card1:2`, which is the shape a driver wants; there are no macros for the
+// halves, so the two sub-fields are spelled out with that citation.
+const clksrc_card0 = Field.of(0, 0x3);
+const clksrc_card1 = Field.of(2, 0x3);
+const cclk_enable = Field.of(regs.SDHOST_CCLK_ENABLE_S, regs.SDHOST_CCLK_ENABLE_V);
+const lp_enable = Field.of(regs.SDHOST_LP_ENABLE_S, regs.SDHOST_LP_ENABLE_V);
+const response_timeout = Field.of(regs.SDHOST_RESPONSE_TIMEOUT_S, regs.SDHOST_RESPONSE_TIMEOUT_V);
+const data_timeout = Field.of(regs.SDHOST_DATA_TIMEOUT_S, regs.SDHOST_DATA_TIMEOUT_V);
+const card_width4 = Field.of(regs.SDHOST_CARD_WIDTH4_S, regs.SDHOST_CARD_WIDTH4_V);
+const card_width8 = Field.of(regs.SDHOST_CARD_WIDTH8_S, regs.SDHOST_CARD_WIDTH8_V);
+const block_size = Field.of(regs.SDHOST_BLOCK_SIZE_S, regs.SDHOST_BLOCK_SIZE_V);
+const byte_count = Field.of(regs.SDHOST_BYTE_COUNT_S, regs.SDHOST_BYTE_COUNT_V);
+const int_mask = Field.of(regs.SDHOST_INT_MASK_S, regs.SDHOST_INT_MASK_V);
+const sdio_int_mask = Field.of(regs.SDHOST_SDIO_INT_MASK_S, regs.SDHOST_SDIO_INT_MASK_V);
+const data_busy = Field.of(regs.SDHOST_DATA_BUSY_S, regs.SDHOST_DATA_BUSY_V);
+const tx_wmark = Field.of(regs.SDHOST_TX_WMARK_S, regs.SDHOST_TX_WMARK_V);
+const rx_wmark = Field.of(regs.SDHOST_RX_WMARK_S, regs.SDHOST_RX_WMARK_V);
+const dma_msize = Field.of(regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_S, regs.SDHOST_DMA_MULTIPLE_TRANSACTION_SIZE_V);
+const bmod_swr = Field.of(regs.SDHOST_BMOD_SWR_S, regs.SDHOST_BMOD_SWR_V);
+const bmod_fb = Field.of(regs.SDHOST_BMOD_FB_S, regs.SDHOST_BMOD_FB_V);
+const bmod_de = Field.of(regs.SDHOST_BMOD_DE_S, regs.SDHOST_BMOD_DE_V);
+const idinten_ti = Field.of(regs.SDHOST_IDINTEN_TI_S, regs.SDHOST_IDINTEN_TI_V);
+const idinten_ri = Field.of(regs.SDHOST_IDINTEN_RI_S, regs.SDHOST_IDINTEN_RI_V);
+const idinten_ni = Field.of(regs.SDHOST_IDINTEN_NI_S, regs.SDHOST_IDINTEN_NI_V);
+
+// The host-side clock generator, which is *not* in the SDMMC block: the P4 moved it into
+// HP_SYS_CLKRST, and it is the first of two divider stages (this one, then CLKDIV inside the
+// controller). `sdmmc_ll.h:227-228` for the source mux and gate, `:244-258` for the divider,
+// `:305-315` for the sampling/driving phase clocks.
+const peri_clk_ctrl01 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL01_REG);
+const peri_clk_ctrl02 = Reg.at(regs.HP_SYS_CLKRST_PERI_CLK_CTRL02_REG);
+
+const sdio_hs_mode = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_S, regs.HP_SYS_CLKRST_REG_SDIO_HS_MODE_V);
+const sdio_ls_clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_SRC_SEL_V);
+const sdio_ls_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EN_V);
+const sdio_ls_clk_edge_cfg_update = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_CFG_UPDATE_V);
+const sdio_ls_clk_edge_l = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_L_V);
+const sdio_ls_clk_edge_h = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_H_V);
+const sdio_ls_clk_edge_n = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_CLK_EDGE_N_V);
+const sdio_ls_slf_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EDGE_SEL_V);
+const sdio_ls_drv_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EDGE_SEL_V);
+const sdio_ls_sam_clk_edge_sel = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EDGE_SEL_V);
+const sdio_ls_slf_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SLF_CLK_EN_V);
+const sdio_ls_drv_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_DRV_CLK_EN_V);
+const sdio_ls_sam_clk_en = Field.of(regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_S, regs.HP_SYS_CLKRST_REG_SDIO_LS_SAM_CLK_EN_V);
+
+// --------------------------------------------------------------------------------- interrupts
+//
+// RINTSTS / INTMASK share one 16-bit layout plus a 2-bit per-card SDIO field at [17:16].
+// `sdmmc_ll.h:35-53` names every bit; the numbers below are those, not a re-derivation.
+
+pub const Event = struct {
+ pub const cd: u32 = 1 << 0; // card detect
+ pub const re: u32 = 1 << 1; // response error
+ pub const cmd_done: u32 = 1 << 2;
+ pub const dto: u32 = 1 << 3; // data transfer over
+ pub const txdr: u32 = 1 << 4;
+ pub const rxdr: u32 = 1 << 5;
+ pub const rcrc: u32 = 1 << 6; // response CRC error
+ pub const dcrc: u32 = 1 << 7; // data CRC error
+ pub const rto: u32 = 1 << 8; // response timeout
+ pub const drto: u32 = 1 << 9; // data read timeout
+ pub const hto: u32 = 1 << 10; // data starvation by host timeout
+ pub const frun: u32 = 1 << 11; // FIFO under/overrun
+ pub const hle: u32 = 1 << 12; // hardware locked write error
+ pub const sbe: u32 = 1 << 13; // RX start-bit error
+ pub const acd: u32 = 1 << 14; // auto command done
+ pub const ebe: u32 = 1 << 15; // end-bit error
+ pub const io_slot0: u32 = 1 << 16;
+ pub const io_slot1: u32 = 1 << 17;
+
+ /// What `sdmmc_ll.h:64-69` (SDMMC_LL_EVENT_DEFAULT) enables at init. Kept exactly as ESP-IDF
+ /// spells it, because the oracle compares against it; what this driver actually unmasks is
+ /// `armed`, below.
+ pub const default: u32 = cd | re | cmd_done | dto | rcrc | dcrc | rto | drto | hto | hle | sbe | ebe;
+
+ /// `default` without card detect, and the only mask `configureInterrupts` ever writes.
+ ///
+ /// Bit 0 has to go, and this is not a preference. There is no card-detect pin on this board:
+ /// `configurePins` ties the signal to a matrix constant 0 ("card present"), and the
+ /// transition it makes while doing so *latches* RINTSTS.cd. RINTSTS is a sticky
+ /// write-1-to-clear register and nothing in the command path clears bit 0 - `sendCommand`
+ /// deliberately writes `default & ~cd` so as not to disturb asynchronous events. So with cd
+ /// unmasked, the controller's single output line into the CLIC is asserted from bring-up
+ /// onwards and never deasserts, and anyone who enables that CLIC line takes an interrupt
+ /// storm that no handler can end. Found by RxPath on CLIC line 21; the fix belongs here
+ /// rather than in the handler, because a level output that nothing can lower is this file's
+ /// bug.
+ ///
+ /// The two SDIO card-interrupt bits are absent from both masks: `setSlaveInterruptEnabled`
+ /// turns the one for this slot on when somebody is prepared to service it.
+ pub const armed: u32 = default & ~cd;
+
+ /// Anything in here means the command failed. `sdmmc_ll.h:71-77` calls the superset
+ /// SDMMC_LL_SD_EVENT_MASK; this is the error half of it.
+ pub const command_errors: u32 = re | rcrc | rto | hle;
+ pub const data_errors: u32 = dcrc | drto | hto | frun | sbe | ebe;
+};
+
+/// The IDMAC's five reportable events - TI, RI, FBE, DU, CES - as one mask. `sdmmc_ll.h:83`
+/// SDMMC_LL_EVENT_DMA_MASK.
+const idsts_event_mask: u32 = 0x1f;
+
+/// The CLIC source this controller raises, for a caller that wants to be woken rather than to
+/// poll. Registering a handler is `hal.intr`'s job and not this file's: see the note on
+/// `slaveInterruptPending`.
+pub const interrupt_source = intr.Source.sdio_host;
+
+// -------------------------------------------------------------------------------------- cache
+//
+// The IDMAC reads its descriptors and its data buffer straight out of L2MEM. The CPU reaches the
+// same L2MEM through the L1 data cache (`soc_caps.h:185`, SOC_CACHE_INTERNAL_MEM_VIA_L1CACHE), and
+// that cache is write-back: `esp_cache_msync(..., DIR_C2M)` exists precisely because a store the
+// CPU has made may still be sitting in a dirty line when the DMA engine reads memory.
+//
+// ESP-IDF offers two ways out and uses both. `sd_trans_sdmmc.c:135-139` writes descriptors through
+// the normal address and calls `esp_cache_msync` after every one. `gdma_link.c:100-118` does it
+// the other way: one write-back-and-invalidate when the region is created, and from then on every
+// CPU access goes through the non-cacheable alias at `addr + 0x40000000`
+// (`hal/cache_ll.h:27` CACHE_LL_L2MEM_NON_CACHE_ADDR, `soc/ext_mem_defs.h:68`).
+//
+// **This file takes the second route.** It is the cheaper one - no cache call in the transfer
+// path - and it is the only one that stays correct without a cache HAL this project does not have.
+// The one-time write-back-and-invalidate is still required, and skipping it is a real bug rather
+// than a theoretical one: `_start` clears .bss with ordinary stores (`src/main.zig:85-92`), so
+// every word of the DMA region below starts life as a *dirty* cache line full of zeros. Nothing
+// says when those lines are evicted; if one is written back after a descriptor has been prepared
+// through the alias, the descriptor becomes zero and the IDMAC stalls on an unowned descriptor.
+// `gdma_link.c:107-112` does exactly this call for exactly this reason.
+//
+// The two ROM entry points are addressed directly rather than declared `extern`, because the
+// generated linker script provides only `ets_printf` and `ets_delay_us`. The addresses are
+// ESP-IDF's, from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld:186` and `:190` - the hw_ver1
+// file, which is the one that matches this die. (If they move into the linker script beside the
+// other two, these two lines become `extern fn` and nothing else changes.)
+
+/// `soc/ext_mem_defs.h:68` SOC_NON_CACHEABLE_OFFSET.
+pub const non_cacheable_offset: u32 = 0x4000_0000;
+
+/// `cache_ll_l1_dcache_get_line_size` reports this on the P4, and `sdmmc_struct.h:36-38` states it
+/// in prose: "On P4, L1 Cache alignment is 64B".
+pub const cache_line: u32 = 64;
+
+/// `rom/cache.h:230` - CACHE_MAP_L1_DCACHE is BIT(4).
+const cache_map_l1_dcache: u32 = 1 << 4;
+
+const romCacheWriteBackAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int =
+ @ptrFromInt(0x4fc0_03f4);
+const romCacheInvalidateAddr: *const fn (map: u32, addr: u32, size: u32) callconv(.c) c_int =
+ @ptrFromInt(0x4fc0_03e4);
+
+// ------------------------------------------------------------------------------- DMA descriptor
+
+/// One IDMAC descriptor, exactly as the hardware reads it: `sdmmc_struct.h:13-41`.
+///
+/// ESP-IDF's `sdmmc_desc_t` is 64 bytes, not 16, and its own comment says why and when not to:
+/// "These `reserved[12]` are for cache alignment... For those who want to access the DMA
+/// descriptor in a non-cacheable way, you can consider remove these `reserved[12]` bytes"
+/// (`sdmmc_struct.h:35-39`). That is this file, so the padding is gone and the descriptor is the
+/// 16 bytes the IDMAC actually fetches.
+pub const Descriptor = extern struct {
+ flags: u32,
+ /// [12:0] buffer1_size, [25:13] buffer2_size.
+ sizes: u32,
+ buffer1: u32,
+ /// Also `buffer2_ptr`; which one it is depends on `second_address_chained`.
+ next: u32,
+
+ pub const disable_int_on_completion: u32 = 1 << 1;
+ pub const last_descriptor: u32 = 1 << 2;
+ pub const first_descriptor: u32 = 1 << 3;
+ pub const second_address_chained: u32 = 1 << 4;
+ pub const end_of_ring: u32 = 1 << 5;
+ pub const card_error_summary: u32 = 1 << 30;
+ pub const owned_by_idmac: u32 = 1 << 31;
+
+ /// `sdmmc_struct.h:43` SDMMC_DMA_MAX_BUF_LEN. `buffer1_size` is 13 bits wide, so 8191 would
+ /// fit; ESP-IDF splits at 4096 and so does the bound below.
+ pub const max_buffer_len: u32 = 4096;
+};
+
+/// Bytes of L2MEM this driver owns, and the whole of its dynamic memory: there is no allocator
+/// here and no allocation anywhere in the transfer path.
+///
+/// 2 KiB of payload is chosen against what sits above: ESP-Hosted's SDIO transport moves at most
+/// one 1600-byte frame plus its 12-byte header per CMD53, and the largest single command this
+/// driver can express in block mode is 4 blocks of 512. Anything larger is split across commands
+/// by `transferChunked`, which is correct for both addressing modes, so the number is a
+/// speed/footprint trade and not a limit.
+pub const bounce_len: u32 = 2048;
+
+/// Descriptor and bounce buffer in one cache-line-aligned region, so the one-time maintenance call
+/// is one call over one range whose base and length are both multiples of 64.
+const DmaRegion = extern struct {
+ desc: Descriptor,
+ _pad: [cache_line - @sizeOf(Descriptor)]u8,
+ buf: [bounce_len]u8,
+};
+
+comptime {
+ std.debug.assert(@sizeOf(Descriptor) == 16);
+ std.debug.assert(@sizeOf(DmaRegion) % cache_line == 0);
+ // One descriptor is enough only while the bounce buffer fits in one. If `bounce_len` ever
+ // grows past 4096 this has to become a ring, and this line is what will say so.
+ std.debug.assert(bounce_len <= Descriptor.max_buffer_len);
+}
+
+/// 2112 bytes: 16 of descriptor, 48 of padding to a cache line, 2048 of payload.
+var dma: DmaRegion align(cache_line) = std.mem.zeroes(DmaRegion);
+
+/// Addresses are `usize` rather than `u32` all the way to the register write. On this target the
+/// two are the same type; on the host, where the arithmetic in these helpers is unit-tested,
+/// `@intCast` of a real 64-bit address would panic before the test could check anything.
+inline fn cachedAddr(p: *const anyopaque) usize {
+ return @intFromPtr(p);
+}
+
+/// The address the *CPU* must use for anything in the DMA region. The hardware gets the cached
+/// address - that is not an inconsistency, it is what ESP-IDF does: `gdma_link.c:268-273` hands
+/// `list->items` to the peripheral and `:159` writes through `list->items_nc`. The alias exists to
+/// change how the CPU's loads and stores are treated, and a bus master is not the CPU.
+inline fn uncachedAddr(p: *const anyopaque) usize {
+ return cachedAddr(p) +% @as(usize, non_cacheable_offset);
+}
+
+/// An address as the 32-bit register field the hardware reads it through.
+inline fn busAddr(p: *const anyopaque) u32 {
+ return @intCast(cachedAddr(p));
+}
+
+inline fn descNc() *volatile Descriptor {
+ return @ptrFromInt(uncachedAddr(&dma.desc));
+}
+
+inline fn bufNc() [*]volatile u8 {
+ return @ptrFromInt(uncachedAddr(&dma.buf));
+}
+
+/// Write back and invalidate the DMA region once, so that no dirty line from `_start`'s .bss clear
+/// can later land on top of what the alias writes. After this, the cached alias of this region is
+/// never touched again by anything in this file.
+fn syncDmaRegionOnce() void {
+ const base = busAddr(&dma);
+ const len: u32 = @sizeOf(DmaRegion);
+ _ = romCacheWriteBackAddr(cache_map_l1_dcache, base, len);
+ _ = romCacheInvalidateAddr(cache_map_l1_dcache, base, len);
+}
+
+// -------------------------------------------------------------------------------------- timing
+//
+// Every wait in this file is bounded, and bounded in time rather than in loop iterations: a spin
+// count is a different number on every optimize level, and this board has no debugger, so a wait
+// that never returns is indistinguishable from a crash.
+//
+// The timebase is the RISC-V `cycle` CSR, the unprivileged shadow of `mcycle`, which is what
+// ESP-IDF itself reads on this part (`rv_utils.h`, because SOC_CPU_HAS_CSR_PC is not defined for
+// the P4) and what `src/soc.zig:116-131` already uses. It is deliberately *not* `hal.systimer`:
+// systimer's `init` pulses the peripheral's reset, which would make the timebase jump under any
+// other user, and `systimer.read` returns null when nothing has brought it up - neither is a
+// property a bus driver should impose on its caller.
+//
+// The CPU clock is whatever the bootloader left, measured at 90 MHz on this board and rated to
+// 400. Deadlines are computed at the 400 MHz *ceiling*, so on real silicon every timeout below is
+// between 1x and 4.4x longer than its nominal microseconds. That is the safe direction: a timeout
+// that fires early would turn a slow card into a spurious failure, and a timeout 4x long still
+// terminates.
+const assumed_cpu_hz_max: u32 = 400_000_000;
+
+inline fn cycleLow() u32 {
+ return asm volatile ("csrr %[r], 0xC00"
+ : [r] "=r" (-> u32),
+ );
+}
+
+/// A bounded wait. 32 bits of cycle counter wrap after 10.7 s at the assumed ceiling, which is an
+/// order of magnitude past the longest deadline here, and the wrapping subtraction is correct
+/// across the wrap anyway.
+const Deadline = struct {
+ start: u32,
+ budget: u32,
+
+ inline fn init(us: u32) Deadline {
+ return .{ .start = cycleLow(), .budget = us *% (assumed_cpu_hz_max / 1_000_000) };
+ }
+
+ inline fn expired(self: Deadline) bool {
+ return (cycleLow() -% self.start) >= self.budget;
+ }
+};
+
+/// `sd_host_private.h:62` SD_HOST_SDMMC_RESET_TIMEOUT_US.
+const reset_timeout_us: u32 = 5_000_000;
+/// `sd_host_private.h:61` SD_HOST_SDMMC_START_CMD_TIMEOUT_US - how long the CIU may take to accept
+/// a command word, which is a bus-side handshake and nothing to do with the card.
+const start_cmd_timeout_us: u32 = 1_000_000;
+/// How long to wait for the card's response after the command has been accepted. The controller
+/// has its own response timeout (TMOUT.response_timeout, 255 card clocks) and raises RTO, so this
+/// only has to cover the case where the controller itself never reports anything.
+const command_done_timeout_us: u32 = 200_000;
+/// Data phase. TMOUT.data_timeout is programmed to 100 ms of card clocks, matching
+/// `sd_host_sdmmc.c:531-533`; this outer bound is twice that.
+const data_done_timeout_us: u32 = 200_000;
+/// How long the card may hold DAT0 low before a new data command.
+const busy_timeout_us: u32 = 500_000;
+
+// ------------------------------------------------------------------------------------- geometry
+
+pub const Width = enum { one, four };
+
+/// Slot 1's pads on this board, and the GPIO-matrix signal each carries.
+///
+/// Slot 0 has a direct IO MUX function and slot 1 does not
+/// (`sdmmc_ll.h:88` SDMMC_LL_SLOT_SUPPORT_GPIO_MATRIX(1) is 1, and `sdmmc_periph.c:37-49` has
+/// -1 for every slot-1 IO MUX pin), so every slot-1 signal is routed through the matrix. The
+/// indices are `gpio_sig_map.h:8-18`, reached here through `regs` rather than written out: the
+/// same discipline `hal/gpio.zig` applies to SIG_GPIO_OUT_IDX, for the same reason.
+pub const Pins = struct {
+ clk: u8,
+ cmd: u8,
+ d0: u8,
+ d1: u8,
+ d2: u8,
+ d3: u8,
+};
+
+/// The ESP32-C6 coprocessor's wiring on this board. CLK 18, CMD 19, D0-D3 = 14/15/16/17.
+pub const c6_pins: Pins = .{ .clk = 18, .cmd = 19, .d0 = 14, .d1 = 15, .d2 = 16, .d3 = 17 };
+
+const sig = struct {
+ const cclk: u32 = @intCast(regs.SD_CARD_CCLK_2_PAD_OUT_IDX);
+ const ccmd: u32 = @intCast(regs.SD_CARD_CCMD_2_PAD_OUT_IDX);
+ const cdata0: u32 = @intCast(regs.SD_CARD_CDATA0_2_PAD_OUT_IDX);
+ const cdata1: u32 = @intCast(regs.SD_CARD_CDATA1_2_PAD_OUT_IDX);
+ const cdata2: u32 = @intCast(regs.SD_CARD_CDATA2_2_PAD_OUT_IDX);
+ const cdata3: u32 = @intCast(regs.SD_CARD_CDATA3_2_PAD_OUT_IDX);
+ const card_detect: u32 = @intCast(regs.SD_CARD_DETECT_N_2_PAD_IN_IDX);
+ const card_int: u32 = @intCast(regs.SD_CARD_INT_N_2_PAD_IN_IDX);
+
+ comptime {
+ // The `_2` in these names is slot 1: `sdmmc_periph.c:52-76` fills
+ // `sdmmc_slot_gpio_sig[1]` from exactly these macros. Slot 0's set is named `_1` and would
+ // route the wrong controller port to the C6's pads, silently.
+ std.debug.assert(cclk == 0 and ccmd == 1 and cdata0 == 2);
+ std.debug.assert(cdata1 == 3 and cdata2 == 4 and cdata3 == 5);
+
+ // The card interrupt is sensed on D1's *input* index, and `configurePins` hands `matrixIn`
+ // the *output* one - correct only because the P4's two signal tables agree on this signal.
+ // `gpio_sig_map.h:13-14` gives cdata1 the number 3 in both directions, and ESP-IDF relies
+ // on the same coincidence: `configure_pin_gpio_matrix` (`sd_host_sdmmc.c:1091-1105`) passes
+ // one `gpio_matrix_sig` to both `esp_rom_gpio_connect_in_signal` and `..._out_signal`.
+ // Asserted rather than assumed, because a mismatch here would route data correctly and
+ // sense interrupts from the wrong pad - which is invisible until something waits.
+ std.debug.assert(cdata1 == @as(u32, @intCast(regs.SD_CARD_CDATA1_2_PAD_IN_IDX)));
+ }
+};
+
+// -------------------------------------------------------------------------------------- state
+
+const State = struct {
+ slot: u1 = 1,
+ width: Width = .four,
+ /// The frequency `cardInit` switches to once the card is addressed and in 4-bit mode.
+ target_khz: u32 = 40_000,
+ pins: Pins = c6_pins,
+ /// Relative card address from CMD3, needed as the argument of CMD7.
+ rca: u16 = 0,
+ initialised: bool = false,
+};
+
+var state: State = .{};
+
+/// The card's relative address, as returned by CMD3. Zero until `cardInit` has run.
+pub fn rca() u16 {
+ return state.rca;
+}
+
+inline fn slotBit() u32 {
+ return @as(u32, 1) << state.slot;
+}
+
+// ------------------------------------------------------------------------------ command words
+//
+// One word, one function, no hardware. This is the part of the driver most worth testing on the
+// host, and the part the oracle can compare against ESP-IDF's own bitfield struct without going
+// anywhere near the card.
+
+/// Compose one field's contribution to a register word. `mmio.Reg.write` does this against a
+/// register; here the destination is a value, because the command word is built, checked and only
+/// then stored.
+inline fn bits(comptime f: Field, v: u32) u32 {
+ return (v & f.unshiftedMask()) << f.shift;
+}
+
+const cmd_index = Field.of(regs.SDHOST_CMD_INDEX_S, regs.SDHOST_CMD_INDEX_V);
+const response_expect = Field.of(regs.SDHOST_RESPONSE_EXPECT_S, regs.SDHOST_RESPONSE_EXPECT_V);
+const response_length = Field.of(regs.SDHOST_RESPONSE_LENGTH_S, regs.SDHOST_RESPONSE_LENGTH_V);
+const check_response_crc = Field.of(regs.SDHOST_CHECK_RESPONSE_CRC_S, regs.SDHOST_CHECK_RESPONSE_CRC_V);
+const data_expected = Field.of(regs.SDHOST_DATA_EXPECTED_S, regs.SDHOST_DATA_EXPECTED_V);
+const read_write = Field.of(regs.SDHOST_READ_WRITE_S, regs.SDHOST_READ_WRITE_V);
+const transfer_mode = Field.of(regs.SDHOST_TRANSFER_MODE_S, regs.SDHOST_TRANSFER_MODE_V);
+const send_auto_stop = Field.of(regs.SDHOST_SEND_AUTO_STOP_S, regs.SDHOST_SEND_AUTO_STOP_V);
+const wait_prvdata_complete = Field.of(regs.SDHOST_WAIT_PRVDATA_COMPLETE_S, regs.SDHOST_WAIT_PRVDATA_COMPLETE_V);
+const stop_abort_cmd = Field.of(regs.SDHOST_STOP_ABORT_CMD_S, regs.SDHOST_STOP_ABORT_CMD_V);
+const send_initialization = Field.of(regs.SDHOST_SEND_INITIALIZATION_S, regs.SDHOST_SEND_INITIALIZATION_V);
+const card_number = Field.of(regs.SDHOST_CARD_NUMBER_S, regs.SDHOST_CARD_NUMBER_V);
+const update_clock_registers_only = Field.of(regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_S, regs.SDHOST_UPDATE_CLOCK_REGISTERS_ONLY_V);
+/// `sdmmc_reg.h:486-494` spells this `USE_HOLE_REG`; `sdmmc_struct.h:473` spells it
+/// `use_hold_reg`, which is what it is - the hold register that synchronises CMD and DATA to
+/// cclk_out. Same bit 29, and ESP-IDF sets it on every command (`sd_host_sdmmc.c:859-860`).
+const use_hold_reg = Field.of(regs.SDHOST_USE_HOLE_REG_S, regs.SDHOST_USE_HOLE_REG_V);
+const start_cmd = Field.of(regs.SDHOST_START_CMD_S, regs.SDHOST_START_CMD_V);
+
+pub const Response = enum { none, short, long };
+pub const Direction = enum { read, write };
+
+/// Everything that distinguishes one command from another, in the terms the register uses.
+pub const Command = struct {
+ index: u6,
+ response: Response = .none,
+ /// Whether the controller checks the response's CRC7. Off for R3 and R4, which do not carry a
+ /// valid one - `sd_protocol_types.h:140-141` define both without SCF_RSP_CRC, and
+ /// `make_hw_cmd` (`sd_trans_sdmmc.c:214-216`) keys `check_response_crc` off exactly that flag.
+ check_crc: bool = false,
+ data: ?Direction = null,
+ /// 80 clocks of 1 before the command. Required once after power-on, and set only for CMD0,
+ /// which is where ESP-IDF sets it (`sd_trans_sdmmc.c:197-206`).
+ send_init: bool = false,
+ /// Wait for a previous data transfer to finish before sending. Set on everything except CMD0,
+ /// CMD12 and CMD11, again following `make_hw_cmd`.
+ wait_prvdata: bool = true,
+ auto_stop: bool = false,
+ stop_abort: bool = false,
+ /// Not a command at all: push CLKDIV/CLKSRC/CLKENA into the card clock domain.
+ update_clock: bool = false,
+ slot: u1 = 0,
+};
+
+/// The 32-bit word that, written to SDHOST_CMD_REG, issues `c`.
+///
+/// This is `make_hw_cmd` (`sd_trans_sdmmc.c:190-229`) plus the three fields
+/// `sd_host_slot_start_command` adds afterwards - `use_hold_reg`, `card_num` and `start_command`
+/// (`sd_host_sdmmc.c:859-881`) - because those three are not optional and splitting them across
+/// two functions is how one of them gets forgotten.
+pub fn commandWord(c: Command) u32 {
+ var w: u32 = 0;
+ w |= bits(cmd_index, c.index);
+ if (c.response != .none) w |= bits(response_expect, 1);
+ if (c.response == .long) w |= bits(response_length, 1);
+ if (c.check_crc) w |= bits(check_response_crc, 1);
+ if (c.data) |dir| {
+ w |= bits(data_expected, 1);
+ if (dir == .write) w |= bits(read_write, 1);
+ }
+ if (c.auto_stop) w |= bits(send_auto_stop, 1);
+ if (c.wait_prvdata) w |= bits(wait_prvdata_complete, 1);
+ if (c.stop_abort) w |= bits(stop_abort_cmd, 1);
+ if (c.send_init) w |= bits(send_initialization, 1);
+ if (c.update_clock) w |= bits(update_clock_registers_only, 1);
+ w |= bits(card_number, c.slot);
+ // Block transfers only; `transfer_mode` selects stream mode, which no SDIO command uses.
+ w |= bits(transfer_mode, 0);
+ w |= bits(use_hold_reg, 1);
+ w |= bits(start_cmd, 1);
+ return w;
+}
+
+// ------------------------------------------------------------------------------ SDIO protocol
+//
+// Command indices and argument layouts, from `sd_protocol_defs.h`. Written out as constants rather
+// than reached through `regs` because they are the SD specification, not this chip: the register
+// headers know nothing about them.
+
+/// `sd_protocol_defs.h:35`, `:40`, `:61`, `:78-80`.
+const cmd_go_idle_state: u6 = 0;
+const cmd_send_relative_addr: u6 = 3;
+const cmd_io_send_op_cond: u6 = 5;
+const cmd_select_card: u6 = 7;
+const cmd_io_rw_direct: u6 = 52;
+const cmd_io_rw_extended: u6 = 53;
+
+/// CMD52's argument: `sd_protocol_defs.h:484-492`.
+pub fn cmd52Arg(write: bool, func: u3, addr: u17, raw_flag: bool, data: u8) u32 {
+ var a: u32 = 0;
+ if (write) a |= @as(u32, 1) << 31;
+ a |= @as(u32, func) << 28;
+ if (raw_flag) a |= @as(u32, 1) << 27;
+ a |= @as(u32, addr) << 9;
+ a |= data;
+ return a;
+}
+
+/// CMD53's argument: `sd_protocol_defs.h:496-506`.
+///
+/// `count` is blocks in block mode and bytes in byte mode, and it is 9 bits: 0 means 512 in byte
+/// mode ("See 5.3.1 SDIO simplified spec", `sdmmc_io.c:351-355`) and infinite in block mode, which
+/// this driver never asks for.
+pub fn cmd53Arg(write: bool, func: u3, addr: u17, block_mode: bool, incrementing: bool, count: u9) u32 {
+ var a: u32 = 0;
+ if (write) a |= @as(u32, 1) << 31;
+ a |= @as(u32, func) << 28;
+ if (block_mode) a |= @as(u32, 1) << 27;
+ if (incrementing) a |= @as(u32, 1) << 26;
+ a |= @as(u32, addr) << 9;
+ a |= count;
+ return a;
+}
+
+/// The block size this driver programmes into BLKSIZ and into the card's CCCR/FBR.
+/// `sdmmc_common.h:195` SDMMC_IO_BLOCK_SIZE, and ESP-Hosted writes the same 512 into FN0 and FN1
+/// (`port_esp_hosted_host_sdio.c:211-217`).
+pub const io_block_size: u32 = 512;
+
+/// CCCR register offsets, `sd_protocol_defs.h:509-530`.
+pub const cccr = struct {
+ pub const revision: u17 = 0x00;
+ pub const fn_enable: u17 = 0x02;
+ pub const fn_ready: u17 = 0x03;
+ pub const int_enable: u17 = 0x04;
+ pub const int_pending: u17 = 0x05;
+ pub const ctl: u17 = 0x06;
+ pub const bus_width: u17 = 0x07;
+ pub const card_cap: u17 = 0x08;
+ pub const cis_ptr: u17 = 0x09;
+ pub const blksize_l: u17 = 0x10;
+ pub const blksize_h: u17 = 0x11;
+
+ pub const ctl_reset: u8 = 1 << 3;
+ pub const bus_width_1: u8 = 0;
+ pub const bus_width_4: u8 = 2;
+ /// Low-speed card; and "4-bit low speed", which says a low-speed card supports 4 bits anyway.
+ pub const card_cap_lsc: u8 = 1 << 6;
+ pub const card_cap_4bls: u8 = 1 << 7;
+};
+
+/// `sd_protocol_defs.h:533` SD_IO_FBR_START - function n's register block starts here.
+const fbr_start: u17 = 0x100;
+
+/// R4's fields, `sd_protocol_defs.h:478-481`.
+const r4_mem_ready: u32 = 1 << 31;
+const r4_mem_present: u32 = 1 << 27;
+
+/// The voltage window the host offers in CMD5's second pass: bits 23:15, i.e. 2.8-3.6 V.
+/// `sd_protocol_defs.h:109` SD_OCR_VOL_MASK, which is the whole of what `get_host_ocr` returns -
+/// "For now tell that the host has 2.8-3.6V voltage range" (`sdmmc_common.h:174-180`).
+const host_ocr: u32 = 0x00ff_8000;
+
+// ------------------------------------------------------------------------------- command issue
+
+/// Write one command word and wait for the CIU to take it. No card traffic is implied: a clock
+/// update command goes through here too.
+///
+/// Both waits are the ones `sd_host_slot_start_command` performs (`sd_host_sdmmc.c:862-892`),
+/// bounded the same way. The first is not redundant with the second: writing any command register
+/// while `start_command` is still set is a hardware locked write error, and HLE is reported
+/// asynchronously in RINTSTS where it is easy to attribute to the wrong command.
+fn startCommand(word: u32, arg: u32) Error!void {
+ var d = Deadline.init(start_cmd_timeout_us);
+ while (cmd.get(start_cmd) != 0) {
+ if (d.expired()) return error.Busy;
+ }
+ cmdarg.writeRaw(arg);
+ cmd.writeRaw(word);
+ d = Deadline.init(start_cmd_timeout_us);
+ while (cmd.get(start_cmd) != 0) {
+ if (d.expired()) return error.Timeout;
+ }
+}
+
+/// Push CLKDIV, CLKSRC and CLKENA into the card clock domain.
+fn clockUpdate() Error!void {
+ try startCommand(commandWord(.{
+ .index = 0,
+ .update_clock = true,
+ .wait_prvdata = true,
+ .slot = state.slot,
+ }), 0);
+}
+
+/// Turn a RINTSTS snapshot into the failure it describes.
+///
+/// Order matters only in that the first match wins, and it is chosen so the most specific cause is
+/// reported: a CRC error and a timeout together is a CRC error, because the timeout is downstream
+/// of it.
+fn decodeErrors(sts: u32) Error!void {
+ if (sts & (Event.rcrc | Event.dcrc) != 0) return error.CrcError;
+ if (sts & (Event.rto | Event.drto | Event.hto) != 0) return error.Timeout;
+ if (sts & (Event.re | Event.hle | Event.ebe | Event.sbe | Event.frun) != 0) return error.ResponseError;
+}
+
+/// Wait for one or more RINTSTS bits, failing on any error bit or on the deadline.
+///
+/// RINTSTS is write-1-to-clear, so this reads with `raw()` and clears with `writeRaw(mask)` -
+/// never `modify`, which would clear every bit it read back and lose the events this function is
+/// not waiting for.
+fn waitEvents(want: u32, errors: u32, us: u32) Error!u32 {
+ const d = Deadline.init(us);
+ while (true) {
+ const sts = rintsts.raw();
+ if (sts & errors != 0) {
+ rintsts.writeRaw(sts & (want | errors));
+ try decodeErrors(sts & errors);
+ // Every bit any caller passes in `errors` is covered above; a new one arriving here
+ // is a bug in this file, and reporting it beats an `unreachable` on a board with no
+ // debugger.
+ return error.ResponseError;
+ }
+ if (sts & want == want) {
+ rintsts.writeRaw(want);
+ return sts;
+ }
+ if (d.expired()) return error.Timeout;
+ }
+}
+
+/// A command with no data phase: issue it, wait for command-done, return R1/R5's first word.
+fn sendCommand(c: Command, arg: u32) Error!u32 {
+ // Everything this command is about to overwrite. This slot's SDIO card interrupt is
+ // deliberately left alone - the C6 raises it asynchronously and clearing it here would drop a
+ // wakeup the layer above is waiting for - and `clearNonSlaveInterrupts` is exactly that set.
+ //
+ // It used to be `Event.default & ~Event.cd`, which is a *subset* of the event bits and left
+ // four of them latched for ever: txdr(4), rxdr(5), frun(11) and acd(14). Two consequences, one
+ // cosmetic and one not. Cosmetic: every RINTSTS a diagnostic prints carries a stale 0x10 from
+ // the first transfer onwards, which is noise in exactly the register that has to be read
+ // carefully. Not cosmetic: **frun is a member of `Event.data_errors`**, so one FIFO
+ // under/overrun - ever - would latch a bit that nothing clears and fail every subsequent
+ // `waitEvents(Event.dto, Event.data_errors, ...)` for the rest of the run. The data path works
+ // today only because frun has never fired.
+ clearNonSlaveInterrupts();
+ var cc = c;
+ cc.slot = state.slot;
+ try startCommand(commandWord(cc), arg);
+ _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us);
+ return resp0.raw();
+}
+
+/// R5's status byte, the one CMD52 and CMD53 return. `sd_protocol_defs.h:493` takes the data byte;
+/// the flags above it say whether the card accepted the command at all.
+const r5_com_crc_error: u32 = 1 << 15;
+const r5_illegal_command: u32 = 1 << 14;
+const r5_error: u32 = 1 << 11;
+const r5_function_number: u32 = 1 << 9;
+const r5_out_of_range: u32 = 1 << 8;
+const r5_bad: u32 = r5_com_crc_error | r5_illegal_command | r5_error | r5_function_number | r5_out_of_range;
+
+fn checkR5(r: u32) Error!u8 {
+ if (r & r5_com_crc_error != 0) return error.CrcError;
+ if (r & r5_bad != 0) return error.ResponseError;
+ return @truncate(r);
+}
+
+// ------------------------------------------------------------------------------- bring-up
+
+/// Controller, FIFO and DMA reset, then wait for all three to self-clear.
+///
+/// All three bits are self-clearing, and `sdmmc_ll.h:486`, `:510` and `:534` each say so with a
+/// different delay ("two AHB clock cycles", "after reset done"). ESP-IDF sets all three and polls
+/// all three together (`sd_host_sdmmc.c:917-950`), which is what makes one bounded wait correct
+/// for the set.
+pub fn resetController() Error!void {
+ ctrl.modify(.{ controller_reset.is(1), fifo_reset.is(1), dma_reset.is(1) });
+ const d = Deadline.init(reset_timeout_us);
+ while (true) {
+ const v = ctrl.raw();
+ if (v & (controller_reset.mask() | fifo_reset.mask() | dma_reset.mask()) == 0) return;
+ if (d.expired()) return error.Timeout;
+ }
+}
+
+/// The interrupt configuration `sd_host_sdmmc.c:120-124` establishes - clear everything, mask
+/// everything, then unmask the completion and error events and turn the global enable on - with
+/// one deliberate deviation: card detect stays masked *and* gets cleared. See `Event.armed` for
+/// why that bit is load-bearing on a board with no card-detect pin.
+///
+/// `int_enable` gates the controller's single line into the CLIC. It is on even though this driver
+/// polls, because RINTSTS is set regardless and the layer above may register a handler for the
+/// SDIO card interrupt; leaving it off would mean `setSlaveInterruptEnabled(true)` silently did
+/// nothing.
+pub fn configureInterrupts() void {
+ rintsts.writeRaw(0xffff_ffff);
+ intmask.writeRaw(0);
+ ctrl.modify(.{int_enable.is(0)});
+ intmask.writeRaw(Event.armed);
+ // Belt and braces: `armed` keeps the controller from reporting a latched cd, and this makes
+ // sure there is no latched cd to report if anything ever unmasks it again.
+ rintsts.writeRaw(Event.cd);
+ ctrl.modify(.{int_enable.is(1)});
+}
+
+/// `sdmmc_ll_init_dma`, `sdmmc_ll.h:796-804`: enable the DMA path, clear the bus-mode register,
+/// pulse the IDMAC's own software reset, and unmask its three completion interrupts.
+pub fn initDma() void {
+ ctrl.modify(.{dma_enable.is(1)});
+ bmod.writeRaw(0);
+ bmod.modify(.{bmod_swr.is(1)});
+ idinten.modify(.{ idinten_ni.is(1), idinten_ri.is(1), idinten_ti.is(1) });
+}
+
+/// Leave the controller's interrupt output silent, and both status registers clean.
+///
+/// `configureInterrupts` and `initDma` above are ESP-IDF's sequences, and ESP-IDF is
+/// interrupt-driven: its transfers wait on a queue its ISR fills, so it needs command-done, the
+/// error bits and the IDMAC's completions in the masks. **This driver polls**, so every one of
+/// those is noise on a line whose only handler understands one cause. Worse than noise: two of
+/// them hold the line asserted forever.
+///
+/// * **INTMASK** gates RINTSTS into MINTSTS. Zero here costs nothing - `waitEvents` reads
+/// RINTSTS, and "Bits are logged regardless of interrupt mask status"
+/// (`sdmmc_struct.h:589-591`).
+/// * **IDINTEN** gates the IDMAC's own events, and it does *not* go through INTMASK. `initDma`
+/// enables NI/RI/TI because `sdmmc_ll_init_dma` does, and IDF can afford that because its ISR
+/// clears IDSTS on every interrupt (`sd_host_sdmmc.c:801-802`). `dataTransfer` clears IDSTS
+/// *before* a transfer and nothing clears it after, so RI and its sticky summary NIS stay set
+/// from the first CMD53 onwards - a permanently asserted interrupt line that no INTMASK write
+/// can lower.
+///
+/// `CTRL.int_enable` stays on: with both masks at zero the line cannot assert anyway, and leaving
+/// the global enable alone keeps `armSlaveInterrupt` down to the stores that matter.
+pub fn muteInterrupts() void {
+ intmask.writeRaw(0);
+ idinten.writeRaw(0);
+ rintsts.writeRaw(0xffff_ffff);
+ idsts.writeRaw(idsts_event_mask);
+}
+
+/// FIFO watermarks and DMA burst size.
+///
+/// ESP-IDF never writes this register on any target - there is no `sdmmc_ll` function for it and
+/// no assignment anywhere in `components/` - so the value in the measured working dump,
+/// `fifoth=0x01FF0000`, is the hardware's reset state: rx watermark 511, tx watermark 0, burst
+/// size code 0 (one transfer). This function writes that value explicitly rather than inheriting
+/// it, because a controller reset is not the only thing that can have touched the register and
+/// "the same as reset" is a claim worth making in code.
+///
+/// It is also a performance knob left deliberately untouched: DesignWare recommends half the FIFO
+/// depth for both watermarks and a burst size matching the AXI port, and tx watermark 0 means a
+/// DMA request only when the FIFO is completely empty. Turning that knob without a board to
+/// measure on would be guessing, and the guess would be against a configuration known to work at
+/// 40 MHz.
+pub fn setFifoThreshold(rx: u32, tx: u32, msize: u32) void {
+ fifoth.write(.{ rx_wmark.is(rx), tx_wmark.is(tx), dma_msize.is(msize) });
+}
+
+/// The reset-value watermarks, which are the ones the working dump shows.
+pub const default_rx_watermark: u32 = 511;
+pub const default_tx_watermark: u32 = 0;
+pub const default_dma_msize: u32 = 0;
+
+/// Bus width, host side. The card side is a CCCR write and is done in `cardInit`; the two must
+/// change in that order, or the next command goes out on a bus the card is not listening to.
+pub fn setBusWidth(w: Width) void {
+ const m = slotBit();
+ const c8 = ctype.get(card_width8) & ~m;
+ const c4 = switch (w) {
+ .one => ctype.get(card_width4) & ~m,
+ .four => ctype.get(card_width4) | m,
+ };
+ ctype.modify(.{ card_width4.is(c4), card_width8.is(c8) });
+}
+
+pub fn setBlockSize(bytes: u32) void {
+ blksiz.modify(.{block_size.is(bytes)});
+}
+
+/// The two-stage divider, resolved. Stage one is `host_div` in HP_SYS_CLKRST, stage two is the
+/// controller's own CLKDIV, and the card clock is `160 MHz / host_div / (2 * card_div)` with
+/// `card_div == 0` meaning bypass.
+///
+/// The table is `sd_host_slot_get_clk_dividers` (`sd_host_sdmmc.c:998-1062`), restricted to the
+/// PLL160M source: this board's C6 is a 3.3 V SDIO device, so the 200 MHz SDIO PLL and the UHS-I
+/// speeds it exists for are out of reach and out of scope.
+pub const Dividers = struct { host: u32, card: u32 };
+
+pub fn dividersFor(khz: u32) Dividers {
+ const src_hz: u32 = 160_000_000;
+ if (khz >= 40_000) return .{ .host = 4, .card = 0 }; // 160/4 = 40 MHz
+ if (khz == 20_000) return .{ .host = 8, .card = 0 }; // 160/8 = 20 MHz
+ if (khz == 400) return .{ .host = 10, .card = 20 }; // 160/10/(20*2) = 400 kHz
+ var host = src_hz / (khz * 1000);
+ var card: u32 = 0;
+ if (host > 15) {
+ host = 2;
+ card = (src_hz / 2) / (2 * khz * 1000);
+ if (((src_hz / 2) % (2 * khz * 1000)) > 0) card += 1;
+ } else if (src_hz % (khz * 1000) > 0) {
+ host += 1;
+ }
+ return .{ .host = host, .card = card };
+}
+
+/// Stage one: the clock generator in HP_SYS_CLKRST. `sdmmc_ll_set_clock_div`,
+/// `sdmmc_ll.h:244-258`.
+///
+/// The `edge_cfg_update` bit is write-to-trigger and must be pulsed - set then cleared - after the
+/// three edge fields, or the new division is programmed and never latched.
+pub fn setHostClockDiv(div: u32) void {
+ if (div > 1) {
+ peri_clk_ctrl02.modify(.{
+ sdio_ls_clk_edge_h.is(div / 2 - 1),
+ sdio_ls_clk_edge_n.is(div - 1),
+ sdio_ls_clk_edge_l.is(div - 1),
+ });
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)});
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)});
+ } else {
+ peri_clk_ctrl01.modify(.{sdio_hs_mode.is(1)});
+ peri_clk_ctrl02.modify(.{
+ sdio_ls_clk_edge_h.is(0),
+ sdio_ls_clk_edge_n.is(0),
+ sdio_ls_clk_edge_l.is(0),
+ });
+ }
+}
+
+/// PLL160M, the only source this driver uses. `sdmmc_ll_select_clk_source`, `sdmmc_ll.h:212-229`:
+/// source value 0 is PLL160M and 1 is the 200 MHz SDIO PLL.
+pub fn selectPll160m() void {
+ peri_clk_ctrl01.modify(.{ sdio_ls_clk_src_sel.is(0), sdio_ls_clk_en.is(1) });
+}
+
+/// The driving, sampling and self clocks the pad logic runs on. `sdmmc_ll_init_phase_delay`,
+/// `sdmmc_ll.h:303-315`. Without this the three gates stay off and the bus does not move, which is
+/// the kind of failure that looks like a wiring fault.
+pub fn initPhaseDelay() void {
+ peri_clk_ctrl02.modify(.{
+ sdio_ls_drv_clk_en.is(1),
+ sdio_ls_sam_clk_en.is(1),
+ sdio_ls_slf_clk_en.is(1),
+ sdio_ls_drv_clk_edge_sel.is(1),
+ sdio_ls_sam_clk_edge_sel.is(0),
+ sdio_ls_slf_clk_edge_sel.is(0),
+ });
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(1)});
+ peri_clk_ctrl02.modify(.{sdio_ls_clk_edge_cfg_update.is(0)});
+}
+
+/// Stage one, whole: divider, source, phase clocks, and the settle the hardware needs afterwards.
+/// `sd_host_set_clk_div`, `sd_host_sdmmc.c:974-990`, including its closing
+/// `esp_rom_delay_us(10)` - "Wait for the clock to propagate".
+///
+/// This has to happen before the controller reset, not after. `controller_reset` is documented to
+/// self-clear "after two AHB and two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`), so with
+/// no card clock reaching the block the bit never clears and the reset wait runs to its full
+/// timeout. ESP-IDF's order says the same thing without saying it: `sd_host_set_clk_div` at
+/// `sd_host_sdmmc.c:109`, `sd_host_reset` at `:112`.
+pub fn setHostClock(div: u32) void {
+ setHostClockDiv(div);
+ selectPll160m();
+ initPhaseDelay();
+ spinMicros(10);
+}
+
+/// Stage two: the controller's per-slot divider and the divider-to-slot mux.
+/// `sdmmc_ll_set_card_clock_div`, `sdmmc_ll.h:431-442`. Slot 1 uses divider 1, slot 0 uses divider
+/// 0 - so the mux value equals the slot number, which is why one line covers both.
+pub fn setCardClockDiv(div: u32) void {
+ if (state.slot == 0) {
+ clksrc.modify(.{clksrc_card0.is(0)});
+ clkdiv.modify(.{clk_divider0.is(div)});
+ } else {
+ clksrc.modify(.{clksrc_card1.is(1)});
+ clkdiv.modify(.{clk_divider1.is(div)});
+ }
+}
+
+/// The card clock's on/off switch, one bit per slot. Takes effect only after a clock update
+/// command. `sdmmc_ll_enable_card_clock`, `sdmmc_ll.h:415-422`.
+pub fn setCardClockEnabled(on: bool) void {
+ const cur = clkena.get(cclk_enable);
+ clkena.modify(.{cclk_enable.is(if (on) cur | slotBit() else cur & ~slotBit())});
+}
+
+/// Stop the card clock while the card is idle. `sdmmc_ll_enable_card_clock_low_power`,
+/// `sdmmc_ll.h:474-481`. **Off** for SDIO: the card raises its interrupt on D1 and cannot do so
+/// with the clock stopped, which is why ESP-IDF clears the same bit for any slot with
+/// `cclk_always_on` (`sd_host_sdmmc.c:272-285`) and why the measured working dump reads
+/// `clkena=0x00000002` rather than `0x00020002`.
+pub fn setCardClockLowPower(on: bool) void {
+ const cur = clkena.get(lp_enable);
+ clkena.modify(.{lp_enable.is(if (on) cur | slotBit() else cur & ~slotBit())});
+}
+
+/// Bytes in the next data transfer. `sdmmc_ll_set_data_transfer_len`, `sdmmc_ll.h:651-654`.
+pub fn setDataTransferLen(len: u32) void {
+ bytcnt.modify(.{byte_count.is(len)});
+}
+
+/// Data-read and response timeouts, both in card output clocks.
+/// `sdmmc_ll_set_data_timeout` / `sdmmc_ll_set_response_timeout`, `sdmmc_ll.h:564-582`.
+pub fn setTimeouts(data_cycles: u32, response_cycles: u32) void {
+ tmout.write(.{
+ data_timeout.is(if (data_cycles > 0xff_ffff) 0xff_ffff else data_cycles),
+ response_timeout.is(response_cycles),
+ });
+}
+
+/// Turn the internal DMA path on or off: both CTRL bits and both BMOD bits, together.
+/// `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`.
+pub fn setDmaEnabled(on: bool) void {
+ const v: u32 = @intFromBool(on);
+ ctrl.modify(.{ dma_enable.is(v), use_internal_dma.is(v) });
+ bmod.modify(.{ bmod_de.is(v), bmod_fb.is(v) });
+}
+
+/// Where the IDMAC fetches its first descriptor. `sdmmc_ll_set_desc_addr`, `sdmmc_ll.h:673-676`.
+/// The address is the *cached* one; see the "Cache" section above for why that is right.
+pub fn setDescriptorAddr(a: u32) void {
+ dbaddr.writeRaw(a);
+}
+
+/// Change the card clock, safely: stop it, reprogramme both stages, start it again, with a clock
+/// update command after each step. `sd_host_slot_set_card_clk`, `sd_host_sdmmc.c:487-537`.
+///
+/// Low-power mode is left **off**, which is the one place this deviates from a plain SD host and
+/// matches the measured dump (`clkena=0x00000002`: clock enabled for slot 1, `lp_enable` clear).
+/// `clkena.lp_enable` stops cclk while the card is idle; an SDIO card signals its interrupt on D1
+/// and needs the clock running to do it, which is why ESP-IDF turns the same bit off for any slot
+/// with `cclk_always_on` (`sd_host_sdmmc.c:272-285`).
+pub fn setBusClock(khz: u32) Error!void {
+ const d = dividersFor(khz);
+
+ setCardClockEnabled(false);
+ try clockUpdate();
+
+ setCardClockDiv(d.card);
+ setHostClock(d.host);
+ try clockUpdate();
+
+ setCardClockEnabled(true);
+ setCardClockLowPower(false);
+ try clockUpdate();
+
+ // 100 ms of card clocks for data, and the maximum 255 card clocks for a response - "always set
+ // response timeout to highest value, it's small enough anyway" (`sd_host_sdmmc.c:534-535`).
+ setTimeouts(100 * khz, 255);
+}
+
+/// Route slot 1's six signals to the C6's pads.
+///
+/// Pull-ups: **the board provides them externally and this enables the internal ones anyway**, on
+/// all six pads, because that is what the working configuration does. It is not obvious from
+/// ESP-Hosted's side - it leaves `SDMMC_SLOT_FLAG_INTERNAL_PULLUP` clear
+/// (`SDMMC_SLOT_CONFIG_DEFAULT`, `sdmmc_default_configs.h:98`: `.flags = 0`) - but every pad still
+/// gets one, because `configure_pin_gpio_matrix` opens with `gpio_reset_pin`
+/// (`sd_host_sdmmc.c:1096`) and that function enables the pull-up unconditionally: "for powersave
+/// reasons, the GPIO should not be floating, select pullup" (`gpio.c:469-472`). The 40 MHz link
+/// that produced the register dump therefore had both the module's external pull-ups and these.
+/// Matching a measured configuration beats reasoning about which resistor is redundant.
+///
+/// D1 has a second job: it is the SDIO interrupt line, and the controller derives that interrupt
+/// from the same routed data signal - `sd_host_slot_sdmmc_io_int_enable` (`sd_host_sdmmc.c:381-388`)
+/// is *only* `configure_pin(d1, sdmmc_slot_gpio_sig[slot].d1, GPIO_MODE_INPUT_OUTPUT)`, the same
+/// two matrix writes and the same `fun_ie` the loop below already does, and it touches no
+/// controller register at all. Both halves are load-bearing and neither is visible in a working
+/// data path: with `fun_ie` clear, or with the *input* side of the matrix left pointing elsewhere,
+/// D1 still drives and every transfer still completes while the controller samples a constant and
+/// never latches a card interrupt. A link that carries traffic and never reports an event is
+/// exactly what that failure looks like, which is why D1 is routed both ways even in 1-bit mode
+/// and why `interruptDiagnostics` prints both bits.
+///
+/// D3 is *not* routed to the controller yet. It is driven high as a plain GPIO output until the
+/// bus is switched to 4 bits, which is how a host tells an SDIO card to use SD mode rather than
+/// SPI mode; `sd_host_sdmmc.c:1282-1294` does the same and `cardInit` reconnects it at
+/// `sd_host_sdmmc.c:575-583`'s point in the sequence.
+pub fn configurePins(pins: Pins) void {
+ // CLK is output-only.
+ gpio.matrixOut(pins.clk, sig.cclk);
+ gpio.setInputEnable(pins.clk, false);
+ gpio.setPull(pins.clk, .up);
+
+ const bidir = [_]struct { pin: u8, signal: u32 }{
+ .{ .pin = pins.cmd, .signal = sig.ccmd },
+ .{ .pin = pins.d0, .signal = sig.cdata0 },
+ .{ .pin = pins.d1, .signal = sig.cdata1 },
+ .{ .pin = pins.d2, .signal = sig.cdata2 },
+ };
+ for (bidir) |b| {
+ gpio.matrixOut(b.pin, b.signal);
+ gpio.matrixIn(b.pin, b.signal);
+ gpio.setInputEnable(b.pin, true);
+ gpio.setPull(b.pin, .up);
+ }
+
+ // D3 high, as a GPIO, until the bus width changes.
+ gpio.configureOutput(pins.d3, .{ .readback = true });
+ gpio.setPull(pins.d3, .up);
+ gpio.setHigh(pins.d3);
+
+ // Card detect and the card's own interrupt-request pin are not wired to anything on this
+ // board, so both are tied off in the matrix exactly as ESP-IDF ties them when no pin is
+ // configured: card-detect to a constant 0 ("card present", `sd_host_sdmmc.c:1315-1319`) and
+ // card-int-n to a constant 1, i.e. inactive (`:1304-1306`). Leaving them unrouted is not the
+ // same thing: GPIO_FUNCn_IN_SEL_CFG resets with `sig_in_sel` clear, which bypasses the matrix
+ // and takes the signal from whatever direct pad function exists - and slot 1 has none.
+ // Write protect is left alone; nothing in this driver reads WRTPRT.
+ gpio.matrixIn(gpio.matrix_const_zero, sig.card_detect);
+ gpio.matrixIn(gpio.matrix_const_one, sig.card_int);
+}
+
+/// Reconnect D3 to the controller, once the card is in 4-bit mode.
+fn attachD3() void {
+ gpio.matrixOut(state.pins.d3, sig.cdata3);
+ gpio.matrixIn(state.pins.d3, sig.cdata3);
+ gpio.setInputEnable(state.pins.d3, true);
+ gpio.setPull(state.pins.d3, .up);
+}
+
+/// Everything from the clock gate to a controller that will accept a command, with the bus at the
+/// 400 kHz probing frequency and 1 bit wide - which is where an SDIO card has to be met.
+///
+/// `cardInit` is what raises it to `khz` and to `width`, after the card has been addressed.
+pub fn init(opts: struct {
+ slot: u1 = 1,
+ width: Width = .four,
+ khz: u32 = 40_000,
+ pins: Pins = c6_pins,
+}) Error!void {
+ state = .{
+ .slot = opts.slot,
+ .width = opts.width,
+ .target_khz = opts.khz,
+ .pins = opts.pins,
+ };
+
+ // The C6 hangs off slot 1 and slot 0's pads are the P4's own flash on most boards; refusing
+ // here is cheaper than debugging a bricked boot.
+ if (opts.slot != 1) return error.NotSupported;
+
+ // 1. Bus clock and reset. Unlike most of this chip, SDMMC's bus clock is gated *off* at
+ // power-on (HP_SYS_CLKRST SOC_CLK_CTRL1 REG_SDMMC_SYS_CLK_EN, default 0), so this is a
+ // prerequisite and not a formality - without it the register block reads stale nonsense.
+ // Its reset bit is not in HP_SYS_CLKRST at all but in LP_AON_CLKRST; see hal/clkrst.zig.
+ clkrst.init(.sdmmc);
+
+ // 2. The host clock generator, *before* the controller reset and not after. `sd_host_reset`
+ // polls three self-clearing bits, and `controller_reset` clears only "after two AHB and
+ // two sdhost_cclk_in clock cycles" (`sdmmc_reg.h:18-20`) - with no card clock reaching the
+ // block that poll runs to its full timeout. ESP-IDF's controller init has the same order:
+ // `sd_host_set_clk_div(ctlr, SDMMC_CLK_SRC_DEFAULT, 2)` at `sd_host_sdmmc.c:109`, then
+ // `sd_host_reset` at `:112`. Divider 2 is IDF's provisional value, replaced at step 6.
+ setHostClock(2);
+
+ // 3. Controller, FIFO and DMA out of reset.
+ try resetController();
+
+ // 4. Interrupts and DMA, before any command can produce one. The first two reproduce ESP-IDF
+ // for the differential; `muteInterrupts` then takes back everything this driver polls for
+ // instead of being interrupted by, leaving the line into the CLIC silent until a waiter
+ // arms it.
+ configureInterrupts();
+ initDma();
+ muteInterrupts();
+
+ // 5. Pads. After the clock gate so the controller's outputs are real, before the card clock so
+ // the first cycle the C6 sees is a clean one.
+ configurePins(state.pins);
+
+ // 6. Bus clock at probing speed, 1 bit wide. An SDIO card has to be met at 400 kHz in 1-bit
+ // mode; `cardInit` raises both once the card has been addressed.
+ try setBusClock(400);
+ setBusWidth(.one);
+
+ // 7. Transfer geometry.
+ setBlockSize(io_block_size);
+ setFifoThreshold(default_rx_watermark, default_tx_watermark, default_dma_msize);
+ setDescriptorAddr(busAddr(&dma.desc));
+
+ // 8. The one cache operation in this driver's life. See the "Cache" section above.
+ syncDmaRegionOnce();
+
+ state.initialised = true;
+
+ // The one claim in this sequence with no differential case behind it, said out loud once, at
+ // the moment it is true. `configureInterrupts` and `initDma` are compared against ESP-IDF on
+ // the die; `muteInterrupts` cannot be - it is a deliberate deviation from IDF's ISR-driven
+ // design, and a reference implementation of our own decision would prove nothing. This line is
+ // the substitute, and it is worth a print because both zeros are load-bearing: a non-zero
+ // idinten here is an interrupt line that no INTMASK write can ever lower.
+ note("MARK SDMMC_INIT intmask=0x%08x idinten=0x%08x expect 0x00000000 and 0x00000000\r\n", .{
+ intmask.raw(), idinten.raw(),
+ });
+}
+
+// ------------------------------------------------------------------------------ card bring-up
+
+/// CMD0, CMD5, CMD3, CMD7, then the CCCR writes that make function 1 usable: the sequence that
+/// takes the C6 from "powered" to "answers CMD52".
+///
+/// The command half follows `sdmmc_card_init` (`sdmmc_init.c:78-133`) restricted to the SDIO path:
+/// `sdmmc_io_reset`, CMD0, `sdmmc_init_io` (CMD5 twice), `sdmmc_init_rca` (CMD3),
+/// `sdmmc_init_select_card` (CMD7). The CCCR half is ESP-Hosted's `hosted_sdio_card_fn_init`
+/// (`port_esp_hosted_host_sdio.c:143-220`) - enable function 1, wait for it to report ready,
+/// unmask its interrupt, switch to 4 bits, set both block sizes to 512 - because that is what this
+/// particular device needs and IDF's generic SDIO init does not do.
+///
+/// The CMD52 that resets the card is allowed to fail. A device that is already out of reset
+/// answers it; one that is not may time out, and `sdmmc_io_reset` (`sdmmc_io.c:66-83`) accepts
+/// exactly that.
+pub fn cardInit() Error!void {
+ if (!state.initialised) return error.NotSupported;
+
+ // CCCR CTL bit 3: I/O reset. Best-effort, as above.
+ cmd52Write(0, cccr.ctl, cccr.ctl_reset) catch {};
+
+ // CMD0 with the 80-clock init sequence and no response.
+ _ = try sendCommand(.{
+ .index = cmd_go_idle_state,
+ .response = .none,
+ .send_init = true,
+ .wait_prvdata = false,
+ }, 0);
+ // SDMMC_GO_IDLE_DELAY_MS (`sdmmc_common.h:34`), which `sdmmc_send_cmd_go_idle_state` waits
+ // out before returning (`sdmmc_cmd.c:114-116`). CMD0 has no response, so there is nothing to
+ // wait *for*: this is the card's own settling time and skipping it makes the next command a
+ // coin toss.
+ spinMicros(20_000);
+
+ // CMD5 with a zero argument asks "are you an IO card, and what voltages do you take"; R4 has
+ // no CRC, hence `check_crc = false` (`sd_protocol_types.h:141`).
+ const probe = try sendCommand(.{
+ .index = cmd_io_send_op_cond,
+ .response = .short,
+ .check_crc = false,
+ }, 0);
+ const functions = (probe >> 28) & 0x7;
+ if (functions == 0) return error.NotSupported; // answered CMD5, but has no IO function
+
+ // CMD5 again with the voltage window, until the card reports ready. 100 attempts is
+ // `sdmmc_io.c:240`; the 10 ms between them is SDMMC_IO_SEND_OP_COND_DELAY_MS
+ // (`sdmmc_common.h:35`), spent here as a bounded spin rather than a scheduler delay.
+ const ocr = host_ocr & probe;
+ var ready = false;
+ var tries: u32 = 0;
+ while (tries < 100) : (tries += 1) {
+ const r = try sendCommand(.{
+ .index = cmd_io_send_op_cond,
+ .response = .short,
+ .check_crc = false,
+ }, ocr);
+ if (r & r4_mem_ready != 0) {
+ ready = true;
+ break;
+ }
+ spinMicros(10_000);
+ }
+ if (!ready) return error.Timeout;
+
+ // CMD3: the card picks its own relative address and returns it in R6[31:16].
+ const r6 = try sendCommand(.{
+ .index = cmd_send_relative_addr,
+ .response = .short,
+ .check_crc = true,
+ }, 0);
+ state.rca = @truncate(r6 >> 16);
+
+ // CMD7 with that address moves the card from stand-by to transfer state. Every CMD52 and
+ // CMD53 after this is addressed to it implicitly.
+ _ = try sendCommand(.{
+ .index = cmd_select_card,
+ .response = .short,
+ .check_crc = true,
+ }, @as(u32, state.rca) << 16);
+
+ // ---- CCCR: function 1 on.
+ const ioe = try cmd52Read(0, cccr.fn_enable);
+ try cmd52Write(0, cccr.fn_enable, ioe | 0x02);
+
+ // Wait for IOR bit 1. ESP-Hosted polls with a 10 ms gap and gives up after SDIO_INIT_MAX_RETRY
+ // (`port_esp_hosted_host_sdio.c:177-192`).
+ var fn_ready = false;
+ tries = 0;
+ while (tries < 100) : (tries += 1) {
+ if ((try cmd52Read(0, cccr.fn_ready)) & 0x02 != 0) {
+ fn_ready = true;
+ break;
+ }
+ spinMicros(10_000);
+ }
+ if (!fn_ready) return error.Timeout;
+
+ // Master interrupt enable plus function 1's, so the C6 can raise D1.
+ const ie = try cmd52Read(0, cccr.int_enable);
+ try cmd52Write(0, cccr.int_enable, ie | 0x01 | 0x02);
+
+ // ---- Bus width: card first, then host, then D3 joins the bus.
+ if (state.width == .four) {
+ const cap = try cmd52Read(0, cccr.card_cap);
+ // "Not a low-speed card" or "a low-speed card that supports 4 bits" - `sdmmc_io.c:182-183`.
+ if ((cap & cccr.card_cap_lsc) == 0 or (cap & cccr.card_cap_4bls) != 0) {
+ try cmd52Write(0, cccr.bus_width, cccr.bus_width_4);
+ setBusWidth(.four);
+ attachD3();
+ } else {
+ state.width = .one;
+ }
+ }
+
+ // ---- Block size 512 for function 0 and function 1, host side and card side.
+ try setCardBlockSize(0, io_block_size);
+ try setCardBlockSize(1, io_block_size);
+ setBlockSize(io_block_size);
+
+ // ---- Finally the target frequency, now that the card is addressed and the bus is wide.
+ try setBusClock(state.target_khz);
+}
+
+/// The 16-bit block size lives in two consecutive byte registers, low half first
+/// (`port_esp_hosted_host_sdio.c:123-141`). Function n's copy is at `0x100 * n + 0x10`.
+fn setCardBlockSize(func: u3, bytes: u16) Error!void {
+ const base: u17 = fbr_start * @as(u17, func);
+ try cmd52Write(0, base + cccr.blksize_l, @truncate(bytes));
+ try cmd52Write(0, base + cccr.blksize_h, @truncate(bytes >> 8));
+}
+
+/// A bounded busy-wait, for the two places the SDIO specification asks for a delay between
+/// retries. Same conservative frequency assumption as `Deadline`, in the same safe direction: on
+/// this 90 MHz die a 10 ms request takes about 44 ms.
+fn spinMicros(us: u32) void {
+ const d = Deadline.init(us);
+ while (!d.expired()) {}
+}
+
+// ------------------------------------------------------------------------------------- CMD52
+
+/// Read one byte from the card's register space.
+///
+/// This is the whole minimal milestone: after `init` and `cardInit`, `cmd52Read(0, 0x00)` reads
+/// CCCR offset 0 and the byte that comes back is the C6 answering.
+pub fn cmd52Read(func: u3, addr: u17) Error!u8 {
+ const r = try sendCommand(.{
+ .index = cmd_io_rw_direct,
+ .response = .short,
+ .check_crc = true,
+ }, cmd52Arg(false, func, addr, false, 0));
+ return checkR5(r);
+}
+
+/// Write one byte. The RAW flag is not set, matching `sdmmc_io_rw_direct` with SD_ARG_CMD52_WRITE
+/// alone (`sdmmc_io.c:187`); `sdmmc_io_write_byte` adds SD_ARG_CMD52_EXCHANGE when it wants the
+/// previous value back, which no caller here does.
+pub fn cmd52Write(func: u3, addr: u17, value: u8) Error!void {
+ const r = try sendCommand(.{
+ .index = cmd_io_rw_direct,
+ .response = .short,
+ .check_crc = true,
+ }, cmd52Arg(true, func, addr, false, value));
+ _ = try checkR5(r);
+}
+
+// ------------------------------------------------------------------------------------- CMD53
+
+/// How one CMD53 is split. Two rules decide it, and both come from ESP-IDF rather than from the
+/// SDIO specification, because both are properties of this controller:
+///
+/// * **Block mode when the length is a whole number of 512-byte blocks**, byte mode otherwise.
+/// In byte mode the count field is bytes and 0 encodes 512 ("See 5.3.1 SDIO simplified spec",
+/// `sdmmc_io.c:351-355`), so one byte-mode command reaches 512 bytes and no further.
+/// * **A byte-mode length of 4 or more must be a multiple of 4.** `sd_trans_sdmmc.c:526-532`
+/// rejects anything else outright, and `sdmmc_io_read_bytes` works around it by splitting:
+/// "host quirk: SDIO transfer with length not divisible by 4 bytes has to be split into two
+/// transfers: one with aligned length, the other one for the remaining 1-3 bytes"
+/// (`sdmmc_io.c:400-419`). So 6 bytes is two commands, 4 then 2, and 3 bytes is one.
+///
+/// A caller that wants the split to be explicit - ESP-Hosted's block path does, because its
+/// addresses increment across the split - can hand over one whole-block chunk at a time and get
+/// exactly one block-mode command per call. A caller that does not can hand over any length.
+const Chunk = struct {
+ block_mode: bool,
+ /// Bytes in this command.
+ len: u32,
+ /// The CMD53 count field: blocks in block mode, bytes in byte mode with 0 meaning 512.
+ count: u9,
+};
+
+fn nextChunk(remaining: u32) Chunk {
+ if (remaining >= io_block_size and remaining % io_block_size == 0) {
+ const max_blocks = bounce_len / io_block_size;
+ var blocks = remaining / io_block_size;
+ if (blocks > max_blocks) blocks = max_blocks;
+ return .{
+ .block_mode = true,
+ .len = blocks * io_block_size,
+ .count = @intCast(blocks),
+ };
+ }
+ var len = remaining;
+ if (len > io_block_size) len = io_block_size;
+ // The 4-byte rule. Below 4 bytes the whole request goes in one command; at or above it, the
+ // aligned part goes first and the 1-3 byte tail becomes the next chunk.
+ if (len >= 4 and len % 4 != 0) len &= ~@as(u32, 3);
+ return .{
+ .block_mode = false,
+ .len = len,
+ .count = if (len == io_block_size) 0 else @intCast(len),
+ };
+}
+
+pub fn cmd53Read(func: u3, addr: u17, buf: []u8, incrementing: bool) Error!void {
+ var offset: u32 = 0;
+ var a: u32 = addr;
+ while (offset < buf.len) {
+ const c = nextChunk(@intCast(buf.len - offset));
+ const arg = cmd53Arg(false, func, @truncate(a), c.block_mode, incrementing, c.count);
+ try dataTransfer(.read, arg, c.len, if (c.block_mode) io_block_size else c.len);
+ const dst = buf[offset..][0..c.len];
+ const src = bufNc();
+ for (dst, 0..) |*b, i| b.* = src[i];
+ offset += c.len;
+ if (incrementing) a += c.len;
+ }
+}
+
+pub fn cmd53Write(func: u3, addr: u17, data: []const u8, incrementing: bool) Error!void {
+ var offset: u32 = 0;
+ var a: u32 = addr;
+ while (offset < data.len) {
+ const c = nextChunk(@intCast(data.len - offset));
+ const src = data[offset..][0..c.len];
+ const dst = bufNc();
+ for (src, 0..) |b, i| dst[i] = b;
+ // The IDMAC moves whole words, so a length that is not a multiple of 4 is rounded up
+ // (`sd_trans_sdmmc.c:127`). Zero the pad rather than send whatever the last transfer left.
+ var pad = c.len;
+ while (pad % 4 != 0) : (pad += 1) dst[pad] = 0;
+ const arg = cmd53Arg(true, func, @truncate(a), c.block_mode, incrementing, c.count);
+ try dataTransfer(.write, arg, c.len, if (c.block_mode) io_block_size else c.len);
+ offset += c.len;
+ if (incrementing) a += c.len;
+ }
+}
+
+/// One CMD53 with its data phase, through the IDMAC and the bounce buffer.
+///
+/// Order is ESP-IDF's (`sd_trans_sdmmc.c:524-568`): descriptor and transfer registers first, then
+/// the command word, then wait for command-done and data-transfer-over in that order. Preparing
+/// the DMA after starting the command would be a race against a card that answers immediately.
+fn dataTransfer(dir: Direction, arg: u32, len: u32, blk: u32) Error!void {
+ std.debug.assert(len <= bounce_len);
+
+ // The card must not still be holding DAT0 low from a previous write.
+ const busy = Deadline.init(busy_timeout_us);
+ while (status.get(data_busy) != 0) {
+ if (busy.expired()) return error.Busy;
+ }
+
+ // As in `sendCommand`: the whole event set except this slot's card interrupt. `frun` is in
+ // `Event.data_errors` and nothing else ever clears it.
+ clearNonSlaveInterrupts();
+ idsts.writeRaw(idsts_event_mask);
+
+ const padded = (len + 3) & ~@as(u32, 3);
+ const d = descNc();
+ d.buffer1 = busAddr(&dma.buf);
+ d.next = 0;
+ d.sizes = padded; // buffer1_size is [12:0]; buffer2 is unused
+ d.flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor |
+ Descriptor.last_descriptor | Descriptor.second_address_chained;
+
+ setDataTransferLen(len);
+ setBlockSize(blk);
+ setDescriptorAddr(busAddr(&dma.desc));
+
+ // `sdmmc_ll_enable_dma`, `sdmmc_ll.h:812-818`, then the poll demand that tells the IDMAC to
+ // re-read a descriptor it may have parked on.
+ setDmaEnabled(true);
+ pldmnd.writeRaw(1);
+
+ try startCommand(commandWord(.{
+ .index = cmd_io_rw_extended,
+ .response = .short,
+ .check_crc = true,
+ .data = dir,
+ .slot = state.slot,
+ }), arg);
+
+ _ = try waitEvents(Event.cmd_done, Event.command_errors, command_done_timeout_us);
+ _ = try checkR5(resp0.raw());
+ _ = try waitEvents(Event.dto, Event.data_errors, data_done_timeout_us);
+}
+
+// -------------------------------------------------------------------- SDIO card interrupt (D1)
+
+/// Has the card asserted its interrupt line?
+///
+/// Non-blocking, no side effect, straight out of RINTSTS bit 16+slot (`sdmmc_reg.h:621-631`). It
+/// does **not** clear the bit; `clearSlaveInterrupt` does, deliberately, once a caller has decided
+/// to act on it.
+///
+/// Two different trigger behaviours meet at this bit and it is worth keeping them apart, because
+/// conflating them sends you tuning the wrong knob:
+///
+/// * **Card -> controller is an edge.** ESP-IDF: "SDIO interrupts are negedge sensitive ones:
+/// the status bit is only set when first interrupt triggered" (`sd_host_sdmmc.c:396-402`).
+/// That is why a waiter must check D1's level once before sleeping - an edge that arrived
+/// while it was awake is not re-delivered.
+/// * **Controller -> CLIC is a level.** RINTSTS is a sticky write-1-to-clear latch, so the
+/// controller's output line stays asserted until software clears the bit that raised it. The
+/// CLIC line therefore wants `.level`, and an edge trigger there would only hide a handler
+/// that fails to deassert rather than fix it.
+///
+/// This is the polling half. The interrupt half is a CLIC line and belongs to whoever owns the
+/// scheduler: route `interrupt_source` with `hal.intr`, and in the handler mask the bit
+/// (`setSlaveInterruptEnabled(false)`) before waking anybody. ESP-IDF does exactly that
+/// (`sd_host_sdmmc.c:826-830`) and explains why at `:396-402`: "SDIO interrupts are negedge
+/// sensitive ones: the status bit is only set when first interrupt triggered", so a handler that
+/// leaves the bit unmasked and unhandled re-enters forever, and a waiter that sleeps without first
+/// checking D1's level loses an edge that arrived while it was awake.
+pub fn slaveInterruptPending() bool {
+ return rintsts.raw() & (Event.io_slot0 << state.slot) != 0;
+}
+
+/// The raw masked-interrupt status word. Diagnostics only: a hang waiting on the card interrupt is
+/// otherwise indistinguishable from a card that never asserted, and this is the register that tells
+/// them apart.
+pub fn interruptStatusRaw() u32 {
+ return rintsts.raw();
+}
+
+pub fn clearSlaveInterrupt() void {
+ rintsts.writeRaw(Event.io_slot0 << state.slot);
+}
+
+/// INTMASK as written. The other half of "why is this line asserted": the controller's output is
+/// RINTSTS AND INTMASK, and a diagnostic that prints only RINTSTS shows half the conjunction.
+///
+/// **Not evidence about an arm.** INTMASK is an ordinary read/write register, so this returns
+/// whatever the last store left - and on the interrupt path the last store is usually the *disarm*.
+/// A caller that wants to know whether unmasking took effect must read the register back inside the
+/// same masked region as the store; that is `armSlaveInterrupt`, and it exists because this
+/// function was read as if it answered that question and it never could.
+pub fn interruptMaskRaw() u32 {
+ return intmask.raw();
+}
+
+/// IDSTS - the IDMAC's own status word, which reaches the controller's interrupt output through
+/// IDINTEN and *not* through INTMASK.
+///
+/// The second independent reason the line can be asserted, and therefore the first thing to read
+/// when a handler entry cannot be explained by RINTSTS. `muteInterrupts` leaves IDINTEN at zero so
+/// this cannot raise the line in this driver; a foreign handler entry with bits set here means
+/// something put IDINTEN back.
+pub fn dmaStatusRaw() u32 {
+ return idsts.raw();
+}
+
+/// Clear every latched event *except* this slot's SDIO card interrupt.
+///
+/// For an interrupt handler that has to lower the controller's output line without racing the
+/// card: the card interrupt is the one event the handler is being woken for, and clearing it here
+/// would drop the wakeup. Everything else - a stale command-done, a latched card-detect, an error
+/// from a transfer that has already been reported - is safe to drop on the floor, and leaving any
+/// of it latched while unmasked keeps the CLIC line high.
+pub fn clearNonSlaveInterrupts() void {
+ rintsts.writeRaw(0x0003_ffff & ~(Event.io_slot0 << state.slot));
+}
+
+/// Unmask this slot's SDIO card interrupt, i.e. let it - and after `muteInterrupts`, *only* it -
+/// reach the CLIC. RINTSTS records the event either way, so polling works without this.
+///
+/// This is the whole of the masking half of arming a waiter: `muteInterrupts` has already left
+/// every other bit of INTMASK and all of IDINTEN at zero, so `true` here makes this slot's card
+/// interrupt the single reason the controller's output can assert - which is what a
+/// level-triggered CLIC line with a one-cause handler requires.
+///
+/// The *order* around it is the part that is easy to get wrong, and it belongs to whoever owns the
+/// scheduler rather than here. `sd_host_slot_sdmmc_io_int_wait` (`sd_host_sdmmc.c:404-426`) is the
+/// reference, and it is four steps:
+///
+/// 1. `setSlaveInterruptEnabled(false)` - mask, so nothing arrives while the state is in flux.
+/// 2. `clearSlaveInterrupt()` - drop the latched edge, so a stale one is not delivered as news.
+/// 3. `slaveInterruptAsserted()` - **if true, act now and do not sleep.** The capture is a
+/// negedge, so with D1 already low step 2 has just thrown away the only edge there will be.
+/// 4. `setSlaveInterruptEnabled(true)` - unmask, and not before. Nothing can be lost between 2
+/// and 4: D1 is a level, and unmasking a bit RINTSTS has already latched asserts the line at
+/// once.
+///
+/// A handler on that line must mask again as its **unconditional first act**, on every path
+/// including the one where the cause turns out not to be its own. The line is a level and it does
+/// not lower itself.
+pub fn setSlaveInterruptEnabled(on: bool) void {
+ // Masked, and that is not decoration. `sdioDispatch` performs *this same* read-modify-write on
+ // *this same* two-bit field, from an interrupt handler, as its unconditional first act. A task
+ // interrupted between the load and the store puts back the bit the handler had just cleared -
+ // re-arming a level-triggered line with nobody left waiting on it, which is precisely how the
+ // storm gets its second chance. Two CSR instructions, and `clkrst.Guard` composes: called from
+ // inside a handler, where MIE is already clear, it leaves MIE clear.
+ //
+ // The register writes are unchanged, so the `sdio_interrupt` differential case still compares
+ // the same resulting word against `sdmmc_ll_enable_sdio_interrupt`'s.
+ const guard = intr.mask();
+ defer guard.release();
+ const m = slotBit();
+ const cur = intmask.get(sdio_int_mask);
+ intmask.modify(.{sdio_int_mask.is(if (on) cur | m else cur & ~m)});
+}
+
+/// What the controller reported the instant after this slot's card interrupt was unmasked.
+pub const Armed = struct {
+ /// The bit the store was trying to set, i.e. `slaveInterruptMask()`.
+ want: u32,
+ /// INTMASK, read back inside the same masked region as the store.
+ intmask: u32,
+ /// MINTSTS - `RINTSTS & INTMASK`, and the only word the controller's output follows. Zero here
+ /// with `stuck()` true is the normal way to enter a sleep: the mask took and nothing is latched
+ /// yet.
+ mintsts: u32,
+ /// RINTSTS, for the case where the edge landed between the unmask and the read-back.
+ rintsts: u32,
+
+ /// Did the unmask take effect?
+ pub inline fn stuck(self: Armed) bool {
+ return self.intmask & self.want != 0;
+ }
+};
+
+/// Unmask this slot's card interrupt and read the result back, both inside one masked region.
+///
+/// This exists because the opposite conclusion was drawn from diagnostics that could not support
+/// it. Every arming window on this board printed `intmask=0x00000000` and that was read as "the
+/// unmask does not stick" - but `MARK PORT_SDIO_LAPSE` prints *after* the disarm, which had just
+/// written that zero deliberately, and `interruptDiagnostics` runs on the application task, which
+/// is never inside an arming window. Neither reading could ever have shown anything else, whatever
+/// the hardware did.
+///
+/// So the claim gets an instrument instead of an argument. Nothing runs between the store and the
+/// three loads: no task, because the runtime is cooperative, and no handler, because MIE is clear.
+/// A `stuck()` of false here is a fact about this register on this die; `stuck()` true retires the
+/// hypothesis.
+pub fn armSlaveInterrupt() Armed {
+ const guard = intr.mask();
+ defer guard.release();
+ const m = slotBit();
+ intmask.modify(.{sdio_int_mask.is(intmask.get(sdio_int_mask) | m)});
+ return .{
+ .want = slaveInterruptMask(),
+ .intmask = intmask.raw(),
+ .mintsts = mintsts.raw(),
+ .rintsts = rintsts.raw(),
+ };
+}
+
+/// This slot's bit in RINTSTS/INTMASK/MINTSTS - the only interrupt cause a waiter here understands.
+pub fn slaveInterruptMask() u32 {
+ return Event.io_slot0 << state.slot;
+}
+
+/// Is the card asserting its interrupt *right now*?
+///
+/// Read from D1's pad rather than from RINTSTS, because the two answer different questions: the
+/// register says "a negedge was latched and not yet cleared", the pad says "the card is holding the
+/// line low". Only the second is safe to test before sleeping, and it is what ESP-IDF tests -
+/// `gpio_get_level(slot_ctx->io_config.d1_io) == 0` at `sd_host_sdmmc.c:413-415`.
+///
+/// Requires D1's input buffer and matrix input to be configured, which `configurePins` does.
+pub fn slaveInterruptAsserted() bool {
+ return gpio.getLevel(state.pins.d1) == 0;
+}
+
+/// The masked status word - what the controller's interrupt output is actually looking at.
+///
+/// Non-zero here and a silent CLIC means the delivery path above the controller is broken (source
+/// routing, line enable, priority, threshold, mstatus.MIE). Zero here while `interruptStatusRaw`
+/// is non-zero means the event is latched but masked, which is the normal resting state of this
+/// driver.
+pub fn interruptStatusMasked() u32 {
+ return mintsts.raw();
+}
+
+// ------------------------------------------------------------------------------- diagnostics
+
+/// `ets_printf` from the mask ROM, the same declaration `src/net/port.zig:75` makes and for the
+/// same reason: this file's only module imports are `regs`, `mmio` and its sibling HALs, and the
+/// symbol comes from the generated linker script rather than from any of them.
+extern fn ets_printf(fmt: [*:0]const u8, ...) c_int;
+
+fn note(comptime fmt: [*:0]const u8, args: anytype) void {
+ _ = @call(.auto, ets_printf, .{fmt} ++ args);
+}
+
+inline fn yesno(b: bool) u32 {
+ return @intFromBool(b);
+}
+
+/// Print the whole card-interrupt delivery chain, in the order a signal traverses it, so that one
+/// flash says which link is broken. Reads registers only: no loop, no wait, no side effect on any
+/// of the state it reports.
+///
+/// The chain has four links and each line below covers one:
+///
+/// * `SDIO_DIAG_PAD` - the card's end. `asserted=1` means D1 is low, i.e. the C6 is requesting
+/// service at this instant. `ie=0` or `in_src` not equal to D1's pad number means the
+/// controller cannot see D1 at all, and no amount of unmasking will help.
+/// * `SDIO_DIAG_TIEOFF` - the two matrix inputs with no pin on this board. `card_int_n` must read
+/// 63 (constant one, inactive) and `card_detect_n` 62 (constant zero, card present), both with
+/// `from_matrix=1`. A `card_int_n` stuck at a constant *zero* is an interrupt that is asserted
+/// before software ever runs, so the first negedge happens before anyone is watching and no
+/// second one ever comes.
+/// * `SDIO_DIAG_CTLR` - the controller's end. `latched` is RINTSTS's bit for this slot,
+/// `unmasked` is INTMASK's, and `mintsts` is the conjunction the interrupt output follows.
+/// `idsts`/`idinten` are the other, independent reason this output can be asserted.
+///
+/// **`unmasked=0` here is the resting state and is not a finding.** This function is called
+/// from an application task; the card interrupt is unmasked only inside an arming window, on
+/// the transport's own task, and is masked again by the handler or the disarm before that task
+/// yields. So an application can never observe the mask up, whatever the hardware does, and
+/// reading a zero here as "the unmask does not stick" is what cost this path a week. The
+/// register read that can answer that question is `armSlaveInterrupt`.
+/// * `SDIO_DIAG_CLIC` - delivery. An unrouted source, a clear `enabled`, a priority at or below
+/// `thresh`, or `mie=0` each mean the line exists and cannot arrive.
+pub fn interruptDiagnostics() void {
+ const m = slaveInterruptMask();
+ const rsts = rintsts.raw();
+ const imask = intmask.raw();
+ const d1 = state.pins.d1;
+ const d1_in = gpio.matrixInSource(sig.cdata1);
+ const ci = gpio.matrixInSource(sig.card_int);
+ const cdet = gpio.matrixInSource(sig.card_detect);
+
+ note("MARK SDIO_DIAG_PAD d1=gpio%u level=%u asserted=%u ie=%u in_src=%u from_matrix=%u inv=%u expect in_src=%u\r\n", .{
+ @as(u32, d1),
+ @as(u32, gpio.getLevel(d1)),
+ yesno(slaveInterruptAsserted()),
+ yesno(gpio.isInputEnabled(d1)),
+ @as(u32, d1_in.pin),
+ yesno(d1_in.from_matrix),
+ yesno(d1_in.inverted),
+ @as(u32, d1),
+ });
+ note("MARK SDIO_DIAG_TIEOFF card_int_n=%u/%u card_detect_n=%u/%u expect 63/1 and 62/1\r\n", .{
+ @as(u32, ci.pin), yesno(ci.from_matrix),
+ @as(u32, cdet.pin), yesno(cdet.from_matrix),
+ });
+ note("MARK SDIO_DIAG_CTLR slot=%u bit=0x%05x latched=%u unmasked=%u rintsts=0x%08x intmask=0x%08x mintsts=0x%08x\r\n", .{
+ @as(u32, state.slot),
+ m,
+ yesno(rsts & m != 0),
+ yesno(imask & m != 0),
+ rsts,
+ imask,
+ interruptStatusMasked(),
+ });
+ note("MARK SDIO_DIAG_CTRL ctrl=0x%08x int_enable=%u idsts=0x%08x idinten=0x%08x status=0x%08x clkena=0x%08x\r\n", .{
+ ctrl.raw(),
+ ctrl.get(int_enable),
+ idsts.raw(),
+ idinten.raw(),
+ status.raw(),
+ clkena.raw(),
+ });
+
+ const src: u32 = @intFromEnum(interrupt_source);
+ if (intr.routedLine(interrupt_source)) |line| {
+ note("MARK SDIO_DIAG_CLIC source=%u line=%u enabled=%u pending=%u trigger=%u prio=%u thresh=%u mie=%u\r\n", .{
+ src,
+ @as(u32, line),
+ yesno(intr.isEnabled(line)),
+ yesno(intr.isPending(line)),
+ @as(u32, @intFromEnum(intr.getTrigger(line))),
+ @as(u32, intr.getPriority(line)),
+ @as(u32, intr.getThreshold()),
+ yesno(intr.globalEnabled()),
+ });
+ } else {
+ note("MARK SDIO_DIAG_CLIC source=%u UNROUTED - no CLIC line can deliver this interrupt\r\n", .{src});
+ }
+}
+
+// --------------------------------------------------------------------------------- host tests
+//
+// Everything below runs on the host under `zig build test`. It covers the two things in this file
+// that are pure functions of their arguments - the command word and the CMD52/CMD53 argument
+// layouts - plus the divider table and the chunking rule. The register sequences are not testable
+// here; that is what src/oracle/sdmmc_cases.zig is for.
+
+const testing = std.testing;
+
+test "CMD52 read is the word ESP-IDF builds" {
+ // make_hw_cmd for {opcode 52, SCF_CMD_AC | SCF_RSP_R5}: response_expect (R5 is PRESENT),
+ // check_response_crc (R5 has CRC), wait_complete (not CMD0/12/11), no data. Then
+ // sd_host_slot_start_command adds use_hold_reg, card_num and start_command.
+ const w = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_2174), w);
+}
+
+test "CMD52 on slot 0 differs from slot 1 only in card_num" {
+ const s0 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 0 });
+ const s1 = commandWord(.{ .index = 52, .response = .short, .check_crc = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA000_2174), s0);
+ try testing.expectEqual(@as(u32, 1 << 16), s0 ^ s1);
+}
+
+test "CMD53 sets data_expected, and rw only when writing" {
+ const rd = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .read, .slot = 1 });
+ const wr = commandWord(.{ .index = 53, .response = .short, .check_crc = true, .data = .write, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_2375), rd);
+ try testing.expectEqual(@as(u32, 0xA001_2775), wr);
+ try testing.expectEqual(@as(u32, 1 << 10), rd ^ wr);
+}
+
+test "CMD0 sends the init sequence and expects nothing back" {
+ // The only command make_hw_cmd gives send_init and denies wait_complete.
+ const w = commandWord(.{ .index = 0, .send_init = true, .wait_prvdata = false, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_8000), w);
+ try testing.expectEqual(@as(u32, 0), w & (1 << 6)); // no response expected
+}
+
+test "CMD5's response CRC is not checked" {
+ // R4 is SCF_RSP_PRESENT alone (sd_protocol_types.h:141) - the OCR response carries no valid
+ // CRC7, and checking it would fail every card.
+ const w = commandWord(.{ .index = 5, .response = .short, .check_crc = false, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA001_2045), w);
+ try testing.expectEqual(@as(u32, 0), w & (1 << 8));
+}
+
+test "CMD3 and CMD7 do check it" {
+ try testing.expectEqual(
+ @as(u32, 0xA001_2143),
+ commandWord(.{ .index = 3, .response = .short, .check_crc = true, .slot = 1 }),
+ );
+ try testing.expectEqual(
+ @as(u32, 0xA001_2147),
+ commandWord(.{ .index = 7, .response = .short, .check_crc = true, .slot = 1 }),
+ );
+}
+
+test "the clock update command sends nothing to the card" {
+ const w = commandWord(.{ .index = 0, .update_clock = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 0xA021_2000), w);
+ try testing.expectEqual(@as(u32, 1 << 21), w & (1 << 21));
+ try testing.expectEqual(@as(u32, 0), w & (1 << 6));
+}
+
+test "a long response sets response_length as well as response_expect" {
+ const w = commandWord(.{ .index = 2, .response = .long, .check_crc = true, .slot = 1 });
+ try testing.expectEqual(@as(u32, 1 << 7), w & (1 << 7));
+ try testing.expectEqual(@as(u32, 1 << 6), w & (1 << 6));
+}
+
+test "every command word starts the command and uses the hold register" {
+ for ([_]Command{
+ .{ .index = 52, .response = .short, .check_crc = true },
+ .{ .index = 53, .response = .short, .check_crc = true, .data = .read },
+ .{ .index = 0, .send_init = true, .wait_prvdata = false },
+ }) |c| {
+ const w = commandWord(c);
+ try testing.expect(w & (1 << 31) != 0);
+ try testing.expect(w & (1 << 29) != 0);
+ }
+}
+
+test "CMD52 argument layout" {
+ // Read CCCR 0x00 on function 0: everything zero.
+ try testing.expectEqual(@as(u32, 0), cmd52Arg(false, 0, 0x00, false, 0));
+ // Write 0x02 to CCCR 0x02 (I/O enable) on function 0.
+ try testing.expectEqual(@as(u32, 0x8000_0402), cmd52Arg(true, 0, 0x02, false, 0x02));
+ // Function 1, address 0x1F800, data 0xAB, with the read-after-write flag.
+ const a = cmd52Arg(true, 1, 0x1F800, true, 0xAB);
+ try testing.expectEqual(@as(u32, 1), a >> 31);
+ try testing.expectEqual(@as(u32, 1), (a >> 28) & 0x7);
+ try testing.expectEqual(@as(u32, 1), (a >> 27) & 1);
+ try testing.expectEqual(@as(u32, 0x1F800), (a >> 9) & 0x1FFFF);
+ try testing.expectEqual(@as(u32, 0xAB), a & 0xFF);
+}
+
+test "CMD53 argument layout, both modes" {
+ // Block mode, function 1, address 0, incrementing, one block.
+ const blk = cmd53Arg(false, 1, 0, true, true, 1);
+ try testing.expectEqual(@as(u32, 0x1C00_0001), blk);
+ // Byte mode, function 1, fixed address, 12 bytes - the ESP-Hosted length read.
+ const byt = cmd53Arg(false, 1, 0x058, false, false, 12);
+ try testing.expectEqual(@as(u32, 0x1000_B00C), byt);
+ // Writing sets bit 31 and nothing else.
+ try testing.expectEqual(
+ @as(u32, 1) << 31,
+ cmd53Arg(true, 1, 0x058, false, false, 12) ^ byt,
+ );
+}
+
+test "byte mode encodes 512 as a count of zero" {
+ // SDIO simplified spec 5.3.1, as applied at sdmmc_io.c:351-355. The chunker never produces
+ // this case - a 512-byte request is a whole block and goes block mode - so the encoder is
+ // checked directly. ESP-Hosted can still reach it: a 512-byte read at a fixed address.
+ try testing.expectEqual(@as(u32, 0), cmd53Arg(false, 1, 0, false, true, 0) & 0x1ff);
+ const c = nextChunk(512);
+ try testing.expect(c.block_mode);
+ try testing.expectEqual(@as(u32, 512), c.len);
+ try testing.expectEqual(@as(u9, 1), c.count);
+}
+
+test "chunking splits on block boundaries and clamps to the bounce buffer" {
+ // Whole blocks, within the buffer: one block-mode command.
+ try testing.expectEqual(@as(u32, 1024), nextChunk(1024).len);
+ try testing.expect(nextChunk(1024).block_mode);
+ // More blocks than fit: clamped to bounce_len, still block mode, still whole blocks.
+ const big = nextChunk(8192);
+ try testing.expect(big.block_mode);
+ try testing.expectEqual(bounce_len, big.len);
+ try testing.expectEqual(@as(u9, bounce_len / 512), big.count);
+ // Not a block multiple: byte mode, count in bytes.
+ const odd = nextChunk(12);
+ try testing.expect(!odd.block_mode);
+ try testing.expectEqual(@as(u32, 12), odd.len);
+ try testing.expectEqual(@as(u9, 12), odd.count);
+ // Longer than one byte-mode command can carry: clamped to 512.
+ const long = nextChunk(1000);
+ try testing.expect(!long.block_mode);
+ try testing.expectEqual(@as(u32, 512), long.len);
+}
+
+test "the controller's 4-byte rule turns a 6-byte transfer into 4 then 2" {
+ // sd_trans_sdmmc.c:526-532 rejects a length that is >= 4 and not a multiple of 4 outright.
+ const first = nextChunk(6);
+ try testing.expect(!first.block_mode);
+ try testing.expectEqual(@as(u32, 4), first.len);
+ const second = nextChunk(6 - first.len);
+ try testing.expectEqual(@as(u32, 2), second.len);
+ // Under four bytes the whole thing goes in one command; that is the case the rule exempts.
+ try testing.expectEqual(@as(u32, 3), nextChunk(3).len);
+ try testing.expectEqual(@as(u32, 1), nextChunk(1).len);
+ // And every chunk a loop produces is either aligned or a final short tail.
+ var remaining: u32 = 1023;
+ var commands: u32 = 0;
+ while (remaining > 0) {
+ const c = nextChunk(remaining);
+ try testing.expect(c.len > 0);
+ try testing.expect(c.len < 4 or c.len % 4 == 0);
+ remaining -= c.len;
+ commands += 1;
+ try testing.expect(commands < 8); // 512 + 508 + 3, not an unbounded walk
+ }
+}
+
+test "divider table reproduces ESP-IDF's three named frequencies" {
+ try testing.expectEqual(Dividers{ .host = 10, .card = 20 }, dividersFor(400));
+ try testing.expectEqual(Dividers{ .host = 8, .card = 0 }, dividersFor(20_000));
+ try testing.expectEqual(Dividers{ .host = 4, .card = 0 }, dividersFor(40_000));
+}
+
+test "divider table lands on or below the requested frequency" {
+ for ([_]u32{ 400, 1_000, 5_000, 10_000, 20_000, 25_000, 40_000 }) |khz| {
+ const d = dividersFor(khz);
+ const div: u64 = @as(u64, d.host) * (if (d.card == 0) @as(u64, 1) else @as(u64, d.card) * 2);
+ const actual_khz = 160_000 / div;
+ try testing.expect(actual_khz <= khz);
+ }
+}
+
+test "the DMA region is one aligned block of exactly the documented size" {
+ try testing.expectEqual(@as(usize, 16), @sizeOf(Descriptor));
+ try testing.expectEqual(@as(usize, 64 + bounce_len), @sizeOf(DmaRegion));
+ try testing.expectEqual(@as(usize, 0), @offsetOf(DmaRegion, "desc"));
+ try testing.expectEqual(@as(usize, 64), @offsetOf(DmaRegion, "buf"));
+ // Both halves of what the one-shot cache maintenance call needs: a base on a cache line and a
+ // length that is a whole number of them. The alignment is on the variable, not on the type -
+ // `@alignOf(DmaRegion)` is 4 - so it has to be checked on the object.
+ try testing.expectEqual(@as(usize, 0), @intFromPtr(&dma) % cache_line);
+ try testing.expectEqual(@as(usize, 0), @sizeOf(DmaRegion) % cache_line);
+}
+
+test "the non-cacheable alias is a fixed offset and nothing more" {
+ var cell: u32 = 0;
+ const a = cachedAddr(&cell);
+ try testing.expectEqual(a +% @as(usize, 0x4000_0000), uncachedAddr(&cell));
+ // And it is the offset ESP-IDF uses, not one this file invented.
+ try testing.expectEqual(@as(u32, 0x4000_0000), non_cacheable_offset);
+}
+
+test "descriptor flags are the bits sdmmc_struct.h names" {
+ try testing.expectEqual(@as(u32, 1 << 2), Descriptor.last_descriptor);
+ try testing.expectEqual(@as(u32, 1 << 3), Descriptor.first_descriptor);
+ try testing.expectEqual(@as(u32, 1 << 4), Descriptor.second_address_chained);
+ try testing.expectEqual(@as(u32, 1 << 31), Descriptor.owned_by_idmac);
+ // The word a single-descriptor transfer writes.
+ const flags = Descriptor.owned_by_idmac | Descriptor.first_descriptor |
+ Descriptor.last_descriptor | Descriptor.second_address_chained;
+ try testing.expectEqual(@as(u32, 0x8000_001C), flags);
+}
+
+test "the default interrupt mask is ESP-IDF's SDMMC_LL_EVENT_DEFAULT" {
+ // sdmmc_ll.h:64-69, expanded: CD|RESP_ERR|CMD_DONE|DATA_OVER|RCRC|DCRC|RTO|DTO|HTO|HLE|SBE|EBE
+ try testing.expectEqual(@as(u32, 0xB7CF), Event.default);
+ // and it deliberately excludes the two per-FIFO-word requests and both SDIO card interrupts.
+ try testing.expectEqual(@as(u32, 0), Event.default & (Event.txdr | Event.rxdr));
+ try testing.expectEqual(@as(u32, 0), Event.default & (Event.io_slot0 | Event.io_slot1));
+}
+
+test "the armed mask drops card detect, and nothing else" {
+ // The bit that produced an unstoppable CLIC line 21: cd latches during pin setup, nothing in
+ // the command path clears it, and while it is unmasked the controller's output never
+ // deasserts. `configureInterrupts` writes `armed`, not `default`.
+ try testing.expectEqual(@as(u32, 0xB7CE), Event.armed);
+ try testing.expectEqual(@as(u32, 0), Event.armed & Event.cd);
+ try testing.expectEqual(Event.cd, Event.default ^ Event.armed);
+ // Every event a transfer actually waits on survives the change.
+ for ([_]u32{ Event.cmd_done, Event.dto, Event.re, Event.rcrc, Event.dcrc, Event.rto, Event.drto, Event.hto, Event.hle, Event.sbe, Event.ebe }) |e| {
+ try testing.expect(Event.armed & e != 0);
+ }
+}