From f5f8068fac59b4f16046c2022c2fc7c7e447ef4c Mon Sep 17 00:00:00 2001 From: Gabriel Schneider Date: Tue, 25 Aug 2026 12:40:53 -0300 Subject: zig-p4: pure-Zig ESP32-P4 toolchain build.zig generates the linker script and drives Zig's own LLD; tools/image.zig turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker. src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers; src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip; src/oracle/ diffs this HAL against ESP-IDF's on the die. --- src/hal/uart.zig | 622 +++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 622 insertions(+) create mode 100644 src/hal/uart.zig (limited to 'src/hal/uart.zig') diff --git a/src/hal/uart.zig b/src/hal/uart.zig new file mode 100644 index 0000000..7c891a2 --- /dev/null +++ b/src/hal/uart.zig @@ -0,0 +1,622 @@ +//! The HP UART controllers: UART0-4. +//! +//! Out of scope on purpose: UHCI/DMA, RS485, IrDA, hardware and software flow control, the wakeup +//! machinery, and LP_UART (which is a different block behind a different clock tree, not an +//! instance of this one). +//! +//! Three things about this peripheral cost real debugging time, and all three are structural rather +//! than incidental: +//! +//! **Half the configuration registers are shadowed.** The registers whose macro name ends `_SYNC` - +//! UART_CLKDIV_SYNC, UART_CONF0_SYNC, and a dozen more - are not the live configuration. A write +//! lands in a shadow that the core clock domain ignores until UART_REG_UPDATE is set, at which +//! point the hardware copies the shadow across and clears the bit itself. Reads come back from the +//! shadow, so a read-modify-write composes correctly and a read-back proves nothing about what the +//! transmitter is currently using. Every mutator here therefore ends in `update()`, which is +//! exactly what ESP-IDF does: `uart_ll_update` (uart_ll.h:85-89) sets the bit and spins on it, and +//! every `*_sync` writer in that file calls it (set_stop_bits at uart_ll.h:793, set_parity at 828, +//! set_data_bit_num at 1023, set_loop_back at 1439, the FIFO resets at 735 and 750). Omitting it +//! does not fail loudly: the register reads back as asked and the wire keeps the old setting. +//! +//! **Reading offset 0x000 pops the RX FIFO.** `UART_FIFO_REG`'s only field is annotated `RO` in +//! uart_reg.h:18 and that annotation is wrong in the way that matters - the read is the pop. A +//! generic "snapshot the block" loop therefore eats received bytes, which is why the differential +//! harness carries a per-peripheral deny-list of offsets. Writes to the same address push a byte, +//! and must be full 32-bit stores: a byte store on this bus is a read-modify-write, so it would pop +//! a byte in order to push one (uart_ll.h:716-724 says so and is the reason `pushByte` uses +//! `writeRaw`). +//! +//! **UART0 is the console.** Resetting it clears UART_CLKDIV, the console turns to garbage +//! mid-sentence and the board dies on a watchdog reset with nothing readable to explain it. That was +//! measured on this board. Nothing here resets UART0 implicitly, `reset()` refuses instance 0, and +//! the differential suite uses UART1. +//! +//! The clock path is two dividers in series and they live in different blocks: HP_SYS_CLKRST holds +//! the integer pre-divider (`REG_UARTn_SCLK_DIV_NUM`) and the source select, the UART itself holds +//! the 12.4 fixed-point divider. `setBaudrate` drives both, because neither alone spans the range. + +const std = @import("std"); +const regs = @import("regs"); +const mmio = @import("mmio"); +const gpio = @import("gpio.zig"); +const clkrst = @import("clkrst.zig"); + +const Reg = mmio.Reg; +const Field = mmio.Field; + +/// UART0-4. LP_UART (ESP-IDF's port 5) is a separate peripheral and not modelled here. +pub const count = 5; + +/// SOC_UART_FIFO_LEN, soc_caps.h:655. Both directions; the TX count register reports how many bytes +/// are queued, so free space is this minus that. +pub const fifo_len = 128; + +// ------------------------------------------------------------------------------ register blocks +// One 0x1000-byte block per instance (soc.h:20, `REG_UART_BASE(i) = DR_REG_UART_BASE + i*0x1000`). +// Every register is reached through a RegArray so the stride is checked against the headers rather +// than assumed, and a wrong instance index is a bounds assert rather than a write into UART2. + +fn regArray(comptime offset: u32) type { + return mmio.RegArray( + regs.DR_REG_UART0_BASE + offset, + regs.DR_REG_UART0_BASE + 0x1000 + offset, + count, + ); +} + +const fifo = regArray(0x00); +const clkdiv_sync = regArray(0x14); +const status = regArray(0x1c); +const conf0_sync = regArray(0x20); +const clk_conf = regArray(0x88); +const reg_update = regArray(0x98); + +// CLKDIV_SYNC: a 12.4 fixed-point divider, with the fraction not adjacent to the integer part. +const clkdiv = Field.of(regs.UART_CLKDIV_S, regs.UART_CLKDIV_V); +const clkdiv_frag = Field.of(regs.UART_CLKDIV_FRAG_S, regs.UART_CLKDIV_FRAG_V); + +// CONF0_SYNC: the data format, the FIFO resets and the loopback switch all share this word, which is +// why every one of them is a read-modify-write and not a `write`. +const parity = Field.of(regs.UART_PARITY_S, regs.UART_PARITY_V); +const parity_en = Field.of(regs.UART_PARITY_EN_S, regs.UART_PARITY_EN_V); +const bit_num = Field.of(regs.UART_BIT_NUM_S, regs.UART_BIT_NUM_V); +const stop_bit_num = Field.of(regs.UART_STOP_BIT_NUM_S, regs.UART_STOP_BIT_NUM_V); +const loopback = Field.of(regs.UART_LOOPBACK_S, regs.UART_LOOPBACK_V); +const rxfifo_rst = Field.of(regs.UART_RXFIFO_RST_S, regs.UART_RXFIFO_RST_V); +const txfifo_rst = Field.of(regs.UART_TXFIFO_RST_S, regs.UART_TXFIFO_RST_V); + +// STATUS: live counters, so read-only and never worth comparing between two runs. +const rxfifo_cnt = Field.of(regs.UART_RXFIFO_CNT_S, regs.UART_RXFIFO_CNT_V); +const txfifo_cnt = Field.of(regs.UART_TXFIFO_CNT_S, regs.UART_TXFIFO_CNT_V); + +const tx_sclk_en = Field.of(regs.UART_TX_SCLK_EN_S, regs.UART_TX_SCLK_EN_V); +const rx_sclk_en = Field.of(regs.UART_RX_SCLK_EN_S, regs.UART_RX_SCLK_EN_V); + +/// UART_REG_UPDATE, the commit bit for the whole `_SYNC` family. `R/W/SC`: the hardware clears it +/// when the copy is done. +const reg_update_bit = Field.of(regs.UART_REG_UPDATE_S, regs.UART_REG_UPDATE_V); + +// -------------------------------------------------------------------------------- clock control +// The source select and the integer pre-divider are one register apart, and not in the register the +// names suggest: for UARTn the select is in PERI_CLK_CTRL(110+n) and the pre-divider is in +// PERI_CLK_CTRL(111+n). That is not a typo in this file - uart_ll.h:463-475 writes +// `peri_clk_ctrl110.reg_uart0_clk_src_sel` while uart_ll.h:558-568 writes +// `peri_clk_ctrl111.reg_uart0_sclk_div_num`, so ctrl111 holds UART0's divider *and* UART1's select. + +const peri_clk_ctrl = mmio.RegArray( + regs.HP_SYS_CLKRST_PERI_CLK_CTRL110_REG, + regs.HP_SYS_CLKRST_PERI_CLK_CTRL111_REG, + 6, // ctrl110..ctrl115: five selects and five dividers, overlapping by one +); + +// All five instances place these fields at the same shifts in their respective registers +// (hp_sys_clkrst_reg.h: every REG_UARTn_CLK_SRC_SEL_S is 24, every REG_UARTn_SCLK_DIV_NUM_S is 0, +// every REG_UARTn_CLK_EN_S is 26), so one macro triple each describes all of them. +const clk_src_sel = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_SRC_SEL_V); +const sclk_div_num = Field.of(regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_UART0_SCLK_DIV_NUM_V); +const sclk_en = Field.of(regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_S, regs.HP_SYS_CLKRST_REG_UART0_CLK_EN_V); + +/// The three clock sources an HP UART can take, with the encoding from uart_ll.h:447-461. +pub const ClockSource = enum(u2) { + /// The 40 MHz crystal. The only source whose frequency is exact, which is why it is the default + /// for anything that has to interoperate. + xtal = 0, + /// RC_FAST, the always-on oscillator. Nominally 20 MHz and uncalibrated - it varies with + /// temperature and part, so a baud rate derived from `nominalHz` here is approximate. + rtc = 1, + /// A fixed 80 MHz tap off the system PLL. Needed for the high rates: the 12-bit integer divider + /// runs out below about 5 kBd from XTAL. + pll_f80m = 2, + + /// The nominal frequency to hand `setBaudrate`. Nominal is exact for `xtal` and `pll_f80m` and a + /// datasheet typical for `rtc`; the real clock tree can be reconfigured, so a caller that has + /// changed it must pass its own number instead. + pub fn nominalHz(self: ClockSource) u32 { + return switch (self) { + .xtal => 40_000_000, + .rtc => 20_000_000, + .pll_f80m => 80_000_000, + }; + } +}; + +pub const WordLength = enum(u2) { + // uart_types.h:57-60. The encoding is (bits - 5), which is why it starts at zero. + bits5 = 0, + bits6 = 1, + bits7 = 2, + bits8 = 3, +}; + +pub const StopBits = enum(u2) { + // uart_types.h:68-70. There is no encoding for zero stop bits, so the enum starts at 1 and 0 is + // reserved by the hardware. + one = 1, + one_and_half = 2, + two = 3, +}; + +pub const Parity = enum(u2) { + // uart_types.h:78-80: bit 1 is "parity enabled", bit 0 is odd/even. `disable` is 0, so the + // odd/even bit is not part of it - see `setParity` for why that matters. + disable = 0, + even = 2, + odd = 3, +}; + +/// One UART instance. A value type holding nothing but the index, so it costs nothing at runtime and +/// every register access folds to a constant address when the index is known. +pub const Uart = struct { + num: u8, + + pub fn init(num: u8) Uart { + std.debug.assert(num < count); + return .{ .num = num }; + } + + // ----------------------------------------------------------------------------- the commit bit + + /// Copy the `_SYNC` shadow registers into the core clock domain and wait for the hardware to + /// acknowledge by clearing the bit (uart_ll.h:85-89). + /// + /// Bounded, where ESP-IDF's `while (hw->reg_update.reg_update);` is not: a UART whose core clock + /// is gated off never clears the bit, and on a board with no debugger an infinite spin is + /// indistinguishable from a crash. 4096 spins is several thousand times the observed cost of a + /// commit, which takes a handful of core-clock cycles. Returns false rather than panicking so a + /// caller can report the peripheral instead of losing the console. + pub fn update(self: Uart) bool { + const r = reg_update.at(self.num); + r.modify(.{reg_update_bit.is(1)}); + return r.waitFor(reg_update_bit, 0, 4096); + } + + // ---------------------------------------------------------------------------- clocks and reset + + /// Reset the block. Refuses UART0. + /// + /// UART0 carries this board's console. A reset clears UART_CLKDIV to its power-on 694, the + /// console's output becomes garbage part-way through whatever it was printing, and the board + /// takes a watchdog reset a moment later - measured, not theorised. There is no "and then put + /// the divider back" version of this that is safe, because the damage is done between the two + /// stores. + pub fn reset(self: Uart) void { + std.debug.assert(self.num != 0); + switch (self.num) { + 1 => clkrst.resetPeripheral(.uart1), + 2 => clkrst.resetPeripheral(.uart2), + 3 => clkrst.resetPeripheral(.uart3), + 4 => clkrst.resetPeripheral(.uart4), + else => unreachable, + } + } + + /// The core (baud-generating) clock, as distinct from the APB bus clock that + /// `clkrst.setClockEnabled` handles. Both are needed: the bus clock makes the registers + /// answer, this one makes the shift registers move - and `update()` is one of the things that + /// stops working without it. + /// + /// Two gates in two blocks, per uart_ll.h:379-397: HP_SYS_CLKRST's per-instance `CLK_EN`, which + /// sits in the *select* register PERI_CLK_CTRL(110+n) and not the divider one next to it, and + /// the UART's own TX and RX enables in UART_CLK_CONF. Interrupts are masked over the first + /// because PERI_CLK_CTRL is shared with unrelated peripherals. + pub fn setCoreClockEnabled(self: Uart, on: bool) void { + const v: u32 = @intFromBool(on); + { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.selectReg().modify(.{sclk_en.is(v)}); + } + clk_conf.at(self.num).modify(.{ tx_sclk_en.is(v), rx_sclk_en.is(v) }); + } + + /// Select the clock the baud generator divides down. Read-modify-write of a register shared with + /// other peripherals, so interrupts are masked (uart_ll.h:477-481 makes the equivalent point by + /// refusing to compile outside `PERIPH_RCC_ATOMIC`). + pub fn setClockSource(self: Uart, src: ClockSource) void { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.selectReg().modify(.{clk_src_sel.is(@intFromEnum(src))}); + } + + pub fn clockSource(self: Uart) ClockSource { + // Encoding 3 is not defined; IDF's getter (uart_ll.h:509-524) maps `default` to RTC, so + // reporting the same thing keeps a round-trip through both implementations consistent. + return switch (self.selectReg().get(clk_src_sel)) { + 0 => .xtal, + 2 => .pll_f80m, + else => .rtc, + }; + } + + /// PERI_CLK_CTRL(110+n): where this instance's source select and core clock gate live. + inline fn selectReg(self: Uart) Reg { + return peri_clk_ctrl.at(self.num); + } + + /// PERI_CLK_CTRL(111+n): where this instance's integer pre-divider lives. One register above + /// the select, which is the trap this pair of accessors exists to contain. + inline fn dividerReg(self: Uart) Reg { + return peri_clk_ctrl.at(self.num + 1); + } + + // ------------------------------------------------------------------------------------- baud + + /// The two dividers a baud rate decomposes into, computed exactly as + /// `_uart_ll_set_baudrate` (uart_ll.h:532-588) does. + pub const Divider = struct { + /// HP_SYS_CLKRST's integer pre-divider, 1-256. Stored as `sclk - 1` in an 8-bit field. + sclk: u32, + /// The UART's own divider, integer part, 12 bits. + int: u32, + /// The UART's own divider, sixteenths. + frag: u32, + }; + + /// Decompose a baud rate, or fail if the hardware cannot express it. + /// + /// The arithmetic, line by line against uart_ll.h: + /// + /// 541 max_div = UART_CLKDIV_V = 0xfff - the UART divider's integer part is 12 bits + /// 542 sclk = ceil(sclk_freq / (max_div * baud)) the smallest pre-divide that brings + /// the remaining ratio inside 12 bits + /// 545 reject sclk == 0 or sclk > 256 256 = SCLK_DIV_NUM_V + 1 + /// 549 clk_div = (sclk_freq << 4) / (baud * sclk) the ratio in sixteenths + /// 551 int = clk_div >> 4 + /// 552 frag = clk_div & 0xf + /// 555+ the field written is sclk - 1 + /// + /// The `<< 4` is IDF's fixed-point scale, not a fudge: CLKDIV_FRAG is a count of sixteenths of a + /// source-clock period added to every bit time, so `clk_div` is the exact ratio rounded down to + /// 1/16 of a tick. Two deliberate departures from the C, neither of which changes a result: + /// + /// * The `ceil` denominator is 64-bit here as it is there (uart_ll.h:542 casts `max_div` to + /// `uint64_t`), and `sclk_freq << 4` is *also* computed in 64 bits. In C that shift is + /// `uint32_t` and overflows above 268.4 MHz; no P4 UART source is anywhere near that (the + /// fastest is PLL_F80M at 80 MHz), so the two agree on every reachable input while this one + /// has no undefined case. + /// * `baud == 0` returns null rather than false-with-registers-untouched; same outcome, but the + /// caller cannot ignore it by accident. + pub fn divider(baud: u32, sclk_freq: u32) ?Divider { + if (baud == 0) return null; + const max_div: u64 = clkdiv.max(); // UART_CLKDIV_V + const denom = max_div * baud; + const sclk: u64 = (@as(u64, sclk_freq) + denom - 1) / denom; + if (sclk == 0 or sclk > @as(u64, sclk_div_num.max()) + 1) return null; + const clk_div: u64 = (@as(u64, sclk_freq) << 4) / (@as(u64, baud) * sclk); + return .{ + .sclk = @intCast(sclk), + .int = @intCast(clk_div >> 4), + .frag = @intCast(clk_div & 0xf), + }; + } + + /// Program a baud rate. Returns false, having touched nothing, if it is unreachable from this + /// source frequency. + /// + /// Store order follows uart_ll.h:550-576 exactly - integer part, fraction, pre-divider, commit - + /// because the intermediate states are visible to the transmitter of a UART that is already + /// running, and because a write-trace comparison against IDF would otherwise differ on ordering + /// while agreeing on the final registers. The two CLKDIV_SYNC stores are separate for the same + /// reason: IDF's two bitfield assignments are two read-modify-writes of that word. + pub fn setBaudrate(self: Uart, baud: u32, sclk_freq: u32) bool { + const d = divider(baud, sclk_freq) orelse return false; + const div = clkdiv_sync.at(self.num); + div.modify(.{clkdiv.is(d.int)}); + div.modify(.{clkdiv_frag.is(d.frag)}); + { + const guard = clkrst.maskInterrupts(); + defer guard.release(); + self.dividerReg().modify(.{sclk_div_num.is(d.sclk - 1)}); + } + _ = self.update(); + return true; + } + + /// The baud rate the registers currently describe, by inverting the above + /// (uart_ll.h:590-615). Integer division both ways, so this is not exactly the value passed to + /// `setBaudrate` - it is what the hardware will actually produce, which is the more useful + /// number. + pub fn baudrate(self: Uart, sclk_freq: u32) u32 { + const div = clkdiv_sync.at(self.num).raw(); + const int = (div >> clkdiv.shift) & clkdiv.unshiftedMask(); + const frag = (div >> clkdiv_frag.shift) & clkdiv_frag.unshiftedMask(); + const sclk = self.dividerReg().get(sclk_div_num) + 1; + const ticks = ((@as(u64, int) << 4) | frag) * sclk; + if (ticks == 0) return 0; + return @intCast((@as(u64, sclk_freq) << 4) / ticks); + } + + // ------------------------------------------------------------------------------ data format + + /// uart_ll.h:1020-1024. + pub fn setWordLength(self: Uart, w: WordLength) void { + conf0_sync.at(self.num).modify(.{bit_num.is(@intFromEnum(w))}); + _ = self.update(); + } + + /// uart_ll.h:790-794. + pub fn setStopBits(self: Uart, s: StopBits) void { + conf0_sync.at(self.num).modify(.{stop_bit_num.is(@intFromEnum(s))}); + _ = self.update(); + } + + /// uart_ll.h:817-832. + /// + /// Note what IDF does *not* do: disabling parity leaves UART_PARITY - the odd/even select bit - + /// at whatever it was, because the value 0 for "disabled" carries no odd/even information and + /// writing bit 0 of it would be writing a zero the caller never asked for. So `.disable` clears + /// `parity_en` only. Reproduced here because otherwise a differential run diverges by one bit + /// after any sequence that sets odd parity and then disables it. + pub fn setParity(self: Uart, p: Parity) void { + const c = conf0_sync.at(self.num); + const v = @intFromEnum(p); + if (p != .disable) c.modify(.{parity.is(v & 1)}); + c.modify(.{parity_en.is((v >> 1) & 1)}); + _ = self.update(); + } + + /// All three format fields, in IDF's order. Three commits rather than one, matching what + /// calling IDF's three setters does: the format of a UART mid-transmission is not atomic on + /// this hardware either way, and diverging here would be a difference with no benefit. + pub fn setFormat(self: Uart, w: WordLength, p: Parity, s: StopBits) void { + self.setWordLength(w); + self.setParity(p); + self.setStopBits(s); + } + + pub fn wordLength(self: Uart) WordLength { + return @enumFromInt(conf0_sync.at(self.num).get(bit_num)); + } + + pub fn stopBits(self: Uart) StopBits { + // Encoding 0 is not a legal stop-bit count. The hardware's reset value is 1, and nothing + // here can write 0, so an out-of-range read means the block is unclocked or was reset + // under us - reported as `one` rather than an illegal enum value, which would be UB. + return switch (conf0_sync.at(self.num).get(stop_bit_num)) { + 2 => .one_and_half, + 3 => .two, + else => .one, + }; + } + + /// uart_ll.h:834-841: parity is only meaningful when enabled, so the odd/even bit is not + /// reported unless it is. + pub fn parityMode(self: Uart) Parity { + const c = conf0_sync.at(self.num).raw(); + if ((c >> parity_en.shift) & 1 == 0) return .disable; + return if ((c >> parity.shift) & 1 == 1) .odd else .even; + } + + // ------------------------------------------------------------------------------------- FIFO + + /// Bytes waiting in the RX FIFO (uart_ll.h:763-766). + pub fn rxCount(self: Uart) u32 { + return status.at(self.num).get(rxfifo_cnt); + } + + /// Bytes queued in the TX FIFO. + pub fn txCount(self: Uart) u32 { + return status.at(self.num).get(txfifo_cnt); + } + + /// Free space in the TX FIFO (uart_ll.h:775-780: the total, minus what is queued). + pub fn txFree(self: Uart) u32 { + return fifo_len - self.txCount(); + } + + /// Push one byte. A full 32-bit store, because a narrower one becomes a read-modify-write on + /// this bus and the read would pop a received byte (uart_ll.h:716-724). + pub inline fn pushByte(self: Uart, byte: u8) void { + fifo.at(self.num).writeRaw(byte); + } + + /// Pop one byte. The read *is* the pop - see this file's header on why offset 0x000 is on the + /// differential harness's no-read list. + pub inline fn popByte(self: Uart) u8 { + return @truncate(fifo.at(self.num).raw()); + } + + /// Discard everything received. Assert, commit, deassert, commit: `rxfifo_rst` lives in a + /// shadow register, so without the commits the hardware never sees either edge + /// (uart_ll.h:733-739). + pub fn resetRxFifo(self: Uart) void { + const c = conf0_sync.at(self.num); + c.modify(.{rxfifo_rst.is(1)}); + _ = self.update(); + c.modify(.{rxfifo_rst.is(0)}); + _ = self.update(); + } + + /// uart_ll.h:748-754. Same shape, and the same reason for it. + pub fn resetTxFifo(self: Uart) void { + const c = conf0_sync.at(self.num); + c.modify(.{txfifo_rst.is(1)}); + _ = self.update(); + c.modify(.{txfifo_rst.is(0)}); + _ = self.update(); + } + + // --------------------------------------------------------------------------------- loopback + + /// Tie TX back to RX inside the block (uart_ll.h:1437-1441). The pads are not involved, which + /// makes it the only way to exercise a UART end to end with nothing wired to the board - it is + /// how the FIFO and format paths can be tested at all here. + pub fn setLoopback(self: Uart, on: bool) void { + conf0_sync.at(self.num).modify(.{loopback.is(@intFromBool(on))}); + _ = self.update(); + } + + pub fn loopbackEnabled(self: Uart) bool { + return conf0_sync.at(self.num).get(loopback) == 1; + } + + // ------------------------------------------------------------------------------ pin routing + + /// This instance's TX signal index in the GPIO matrix. The names in IDF's map are + /// `UARTn_TXD_PAD_OUT_IDX` (gpio_sig_map.h:28-52) and they are consecutive in steps of three, + /// but the step is not relied on: each is named. + pub fn txSignal(self: Uart) u32 { + return switch (self.num) { + 0 => regs.UART0_TXD_PAD_OUT_IDX, + 1 => regs.UART1_TXD_PAD_OUT_IDX, + 2 => regs.UART2_TXD_PAD_OUT_IDX, + 3 => regs.UART3_TXD_PAD_OUT_IDX, + 4 => regs.UART4_TXD_PAD_OUT_IDX, + else => unreachable, + }; + } + + /// This instance's RX signal index. Numerically equal to the TX one - the matrix's input and + /// output signal spaces are separate namespaces that happen to share indices for a duplex + /// peripheral - which is exactly why routing RX with `matrixOut` silently does nothing useful. + pub fn rxSignal(self: Uart) u32 { + return switch (self.num) { + 0 => regs.UART0_RXD_PAD_IN_IDX, + 1 => regs.UART1_RXD_PAD_IN_IDX, + 2 => regs.UART2_RXD_PAD_IN_IDX, + 3 => regs.UART3_RXD_PAD_IN_IDX, + 4 => regs.UART4_RXD_PAD_IN_IDX, + else => unreachable, + }; + } + + /// Route TX to a pad through the GPIO matrix. + pub fn routeTx(self: Uart, pin: u8) void { + gpio.matrixOut(pin, self.txSignal()); + } + + /// Route a pad to RX through the GPIO matrix, and enable that pad's input buffer - without + /// which the routed signal reads as a constant and the UART receives nothing, which is the + /// single most common way this goes wrong. + pub fn routeRx(self: Uart, pin: u8) void { + gpio.setInputEnable(pin, true); + gpio.matrixIn(pin, self.rxSignal()); + } + + // --------------------------------------------------------------------------------- transfers + + /// Send every byte, blocking until each fits. Bounded only by the FIFO draining, which always + /// progresses while the core clock is on - so unlike a blocking *read* this cannot wait on an + /// event that may never happen. + pub fn write(self: Uart, bytes: []const u8) void { + for (bytes) |b| { + while (self.txFree() == 0) {} + self.pushByte(b); + } + } + + /// Drain up to `buf.len` received bytes and report how many there were. Does not block. + /// + /// Deliberately not blocking: nothing on the other end of a UART is obliged to send, so a + /// blocking read is an unbounded wait, and there is no timer in this HAL's dependency set to + /// bound it with. A caller that wants to wait writes the loop, and owns the decision about what + /// to do when the bytes never come. + pub fn read(self: Uart, buf: []u8) usize { + var n: usize = 0; + const available = self.rxCount(); + while (n < buf.len and n < available) : (n += 1) buf[n] = self.popByte(); + return n; + } + + /// Whether the transmitter has finished: nothing queued in the FIFO. + /// + /// Not the same as "the last bit is on the wire" - the shift register still holds up to one + /// character after the FIFO empties. UART_FSM_STATUS reports that, and this HAL does not model + /// it, so a caller about to cut the clock or reconfigure the format must allow for one more + /// character time. + pub fn txIdle(self: Uart) bool { + return self.txCount() == 0; + } +}; + +// ------------------------------------------------------------------------------------ host tests +// The divider arithmetic is the only part of this file that can be checked without the chip, and it +// is the part most worth checking: every value below is IDF's formula evaluated by hand, so a +// transcription error in `divider` fails here rather than as a garbled console. + +test "40 MHz XTAL, 115200 Bd: one source tick, 12.4 divider does the work" { + // 40e6/(4095*115200) = 0.085 -> ceil = 1. clk_div = (40e6<<4)/115200 = 5555 (5555.55 floored). + // 5555 = 347*16 + 3. + const d = Uart.divider(115200, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 1), d.sclk); + try std.testing.expectEqual(@as(u32, 347), d.int); + try std.testing.expectEqual(@as(u32, 3), d.frag); + // 347 + 3/16 = 347.1875 ticks per bit -> 115,213 Bd in real arithmetic, and 115,211 as the + // hardware's own truncating inverse reports it (see the round-trip test): 0.01% fast either way. +} + +test "80 MHz PLL, 115200 Bd: the fraction differs from the XTAL case, which is the point of it" { + // 80e6/(4095*115200) = 0.17 -> 1. clk_div = (80e6<<4)/115200 = 11111 = 694*16 + 7. + const d = Uart.divider(115200, 80_000_000).?; + try std.testing.expectEqual(@as(u32, 1), d.sclk); + try std.testing.expectEqual(@as(u32, 694), d.int); + try std.testing.expectEqual(@as(u32, 7), d.frag); +} + +test "a rate low enough to need the pre-divider" { + // 300 Bd from 40 MHz: 40e6/300 = 133,333 ticks per bit, far past 12 bits. + // ceil(40e6/(4095*300)) = ceil(32.6) = 33. clk_div = (40e6<<4)/(300*33) = 64,646 = 4040*16 + 6. + const d = Uart.divider(300, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 33), d.sclk); + try std.testing.expectEqual(@as(u32, 4040), d.int); + try std.testing.expectEqual(@as(u32, 6), d.frag); + try std.testing.expect(d.int <= 0xfff); + try std.testing.expect(d.sclk <= 256); +} + +test "unreachable rates are rejected rather than rounded" { + // Zero is IDF's explicit early return (uart_ll.h:538). + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(0, 40_000_000)); + // 10 Bd from 40 MHz needs a pre-divide of ceil(40e6/40950) = 977, past the 8-bit field's 256. + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(10, 40_000_000)); +} + +test "the sclk == 0 rejection is unreachable except from a zero source frequency" { + // Worth pinning down, because the obvious reading of uart_ll.h:545 is wrong. `sclk` is a + // *ceiling*, so for any non-zero source frequency it is at least 1 - asking for 4 MBd from a + // 1 kHz clock does NOT fail here, it yields sclk = 1 and a divider of zero, and IDF programs + // that just as happily. The only input that trips the branch is sclk_freq == 0. + const absurd = Uart.divider(4_000_000, 1000).?; + try std.testing.expectEqual(@as(u32, 1), absurd.sclk); + try std.testing.expectEqual(@as(u32, 0), absurd.int); + try std.testing.expectEqual(@as(u32, 0), absurd.frag); + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(115200, 0)); +} + +test "the pre-divider field stores sclk - 1, so the reachable rates stop at 256 ticks" { + // The boundary IDF checks at uart_ll.h:545: sclk may be 256 because the field holds sclk-1. + // From 40 MHz the last rate inside it is 39 Bd, at a pre-divide of 251; 38 Bd needs 257. + const ok = Uart.divider(39, 40_000_000).?; + try std.testing.expectEqual(@as(u32, 251), ok.sclk); + try std.testing.expectEqual(@as(u32, 4086), ok.int); + try std.testing.expectEqual(@as(?Uart.Divider, null), Uart.divider(38, 40_000_000)); +} + +test "the divider round-trips through the baud rate the hardware will really produce" { + // What `baudrate()` computes, without a chip: the inverse of the same arithmetic. + const d = Uart.divider(115200, 40_000_000).?; + const ticks = ((@as(u64, d.int) << 4) | d.frag) * d.sclk; + const actual: u32 = @intCast((@as(u64, 40_000_000) << 4) / ticks); + // 347 + 3/16 = 347.1875 ticks per bit, and 40e6*16/5555 truncates to 115,211 Bd: 0.01% fast. + try std.testing.expectEqual(@as(u32, 115_211), actual); +} -- cgit v1.3