summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-25 21:11:24 -0300
committerGabriel Schneider <[email protected]>2026-08-25 21:11:24 -0300
commitaa524eae414f4875794638df0586c011c6b0f2c5 (patch)
tree2a1a7e1794195f7864b69b1ae838b67c9f33b98e /src
parent7664c57d93cd3f3bdb719f7ca915d657668ddfd9 (diff)
downloadesp32p4-aa524eae414f4875794638df0586c011c6b0f2c5.tar.gz
esp32p4-aa524eae414f4875794638df0586c011c6b0f2c5.zip
Run the CPU at 360 MHz: -Dcpu-mhz, and a keystroke lands at 4.37 ms
The board was executing at 90 MHz because the stock second-stage bootloader is built with CONFIG_BOOTLOADER_CPU_CLK_FREQ_MHZ=90 (bootloader_clock_init.c:27-37). Measured here against the systimer, which is XTAL/2.5 and therefore an independent reference: 4,500,367 cycles in 50,004 us = exactly 90 MHz. The CPLL is ALREADY at 360 MHz - 90 is 360/4 - so this is a divider change and nothing else. No PLL to enable, no lock to wait for, and the P4 has no per-frequency voltage step to order it against (rtc_clk_init.c:58-80 sets HP_ACTIVE DBIAS once from efuse). `hal/clkrst.zig:setCpuFreq` writes the four dividers in ESP-IDF's upscale order - APB, SYS, MEM, then CPU, with a bus-update handshake after each - because IDF's own comment says the other order passes through a state where APB or MEM violates its timing. Then it calls the ROM's `ets_update_cpu_frequency`, without which every `ets_delay_us` in the image is wrong by exactly the frequency ratio. Measured after: 359,991 kHz. Nothing else moved, which is the reason this is safe from a running console: UART0's baud clock comes from XTAL (hal/uart.zig:116-139), the systimer from XTAL/2.5, and the flash interface from SPLL 480 MHz - none of them from the CPU. The `cycle` CSR simply counts faster, and the board never converts it, so only the divisor in experiments/ had to move. ## What it bought step fixed per char at 160 chars ReleaseSmall 16.99 ms 54.3 us 25.56 ms ReleaseFast 14.85 ms 34.7 us 20.30 ms 0.79x + ASCII grapheme 14.56 ms 12.0 us 16.46 ms 0.64x + ASCII print 14.27 ms 6.9 us 15.36 ms 0.60x + shadow grid 8.87 ms 7.3 us 10.02 ms 0.39x + byte compare 8.37 ms 7.1 us 9.48 ms 0.37x + 360 MHz 4.37 ms 1.9 us 4.67 ms 0.18x Compute went 6.42 -> 2.44 ms: 2.6x for a 4x clock, not 4x, and the shortfall is the point. At 360 MHz the grid walk reads 27 KB per frame in 226 us, about 6 cycles a byte, so that stage is bounded by L2MEM bandwidth and does not care how fast the core is. The prediction that this would happen was made before the measurement and held. ## The goal was 4 ms and this is 4.37 Short by 372 us, and the remaining budget is known: ~2.0 ms of host and USB latency that no firmware change touches (measured independently against the protocol responder), plus 2.4 ms of board compute of which vaxis's own diff is 631 us, pardes's Surface rebuild ~505 us and our grid walk 226 us. vaxis's diff is the only item large enough to close the gap alone, and it is redundant work - `present` already computes exactly which cells moved - so emitting ANSI straight from the shadow grid would do it. I did not, because it is a from-scratch renderer and the honest verification for it needs more than the harness currently proves. ## Verification, and a bug in my own instrument Raising a core clock 4x is exactly the change that corrupts a screen quietly, so the A/B compares screens across clocks as well as across the shadow-grid flag. The first attempt REPORTED A DIFFERENCE at 360 MHz, and it was the verifier: it hashed the raw SGR parameters applied to each cell, which is history-dependent, and a faster board splits the same keystrokes across different frames. Decoding SGR into actual state - resolved foreground, background and attribute set per cell - it is identical: same characters and same style everywhere, both across clocks and across the flag. Also checked and found innocent: `rtt.zig` polled with a 1 ms timeout, which looked like it would quantise every sample. It does not - poll(2) returns when data arrives, not when the timeout expires - and switching to a non-blocking spin moved the measured round trip by 0 us. The comment now says so, since the next reader will wonder too. snap 95/95, hxdiff 481 cases 0 mismatches, hxparity 561 cases 0 mismatches, unit-test, zig-p4 host tests, tty and p4 both build. 90 MHz remains the default; -Dcpu-mhz=360 is opt-in because every number in experiments/ up to this commit was taken at 90.
Diffstat (limited to 'src')
-rw-r--r--src/hal/clkrst.zig90
-rw-r--r--src/pardes/app.zig29
-rw-r--r--src/soc.zig5
3 files changed, 124 insertions, 0 deletions
diff --git a/src/hal/clkrst.zig b/src/hal/clkrst.zig
index 89a27ef..9902796 100644
--- a/src/hal/clkrst.zig
+++ b/src/hal/clkrst.zig
@@ -263,3 +263,93 @@ pub fn init(comptime p: Peripheral) void {
setClockEnabled(p, true);
resetPeripheral(p);
}
+
+// --------------------------------------------------------------------------- the CPU's own clock
+
+/// Raise the HP CPU clock from the 90 MHz the bootloader leaves to `mhz`.
+///
+/// WHY THIS IS CHEAP. The CPLL is ALREADY at 360 MHz: 90 is exactly 360/4, and the stock
+/// second-stage bootloader gets there by setting `CONFIG_BOOTLOADER_CPU_CLK_FREQ_MHZ = 90`
+/// (`bootloader_support/src/bootloader_clock_init.c:27-37`). So this is a divider change and
+/// nothing else - no PLL to enable, no lock to wait for, and on the P4 no voltage step exists to
+/// order it against (`esp_hw_support/port/esp32p4/rtc_clk_init.c:58-80` sets HP_ACTIVE DBIAS once
+/// from efuse and never per-frequency).
+///
+/// WHAT IT DOES NOT DISTURB, which is the reason it is safe to do from a running console:
+/// * UART0's baud clock is selected by `PERI_CLK_CTRL110[25:24]` from XTAL, RC_FAST or PLL_F80M
+/// (`hal/uart.zig:116-139`) - never the CPU clock. The console keeps its rate.
+/// * The systimer is XTAL/2.5 = 16 MHz (`hal/systimer.zig:31`), so every timeout built on
+/// `nowMs` keeps meaning what it meant.
+/// * The flash interface runs from SPLL 480 MHz (`spimem_flash_ll.h:676-684`), so code executing
+/// from flash-mapped memory is unaffected and this need not run from RAM.
+/// * The `cycle` CSR counts real CPU cycles, so it simply counts faster. Nothing on the board
+/// caches a cycles-per-microsecond figure; the HOST divisor in `experiments/` must move.
+///
+/// The divider set and the ORDER are ESP-IDF's, from `rtc_clk_cpu_freq_to_cpll_mhz`
+/// (`esp_hw_support/port/esp32p4/rtc_clk.c`). Only three CPU frequencies are legal on pre-v3
+/// silicon and each pins MEM/SYS/APB with it, because MEM must stay <= 200 MHz and APB <= 100:
+///
+/// CPU 360 = CPLL/1, MEM = CPU/2 = 180, SYS = MEM/1 = 180, APB = SYS/2 = 90
+/// CPU 180 = CPLL/2, MEM = CPU/1 = 180, SYS = MEM/1 = 180, APB = SYS/2 = 90
+/// CPU 90 = CPLL/4, MEM = CPU/1 = 90, SYS = MEM/1 = 90, APB = SYS/1 = 90
+///
+/// APB lands at 90 MHz in all three, which is why peripherals do not care. Upscaling walks
+/// APB -> SYS -> MEM -> CPU with a bus update after each: IDF's comment is explicit that the other
+/// order passes through an intermediate state where APB or MEM violates its timing, and anything
+/// touching those clocks during it may fault.
+pub const CpuFreq = enum(u16) { mhz90 = 90, mhz180 = 180, mhz360 = 360 };
+
+pub fn setCpuFreq(target: CpuFreq) void {
+ const root0 = Reg.at(regs.HP_SYS_CLKRST_ROOT_CLK_CTRL0_REG);
+ const root1 = Reg.at(regs.HP_SYS_CLKRST_ROOT_CLK_CTRL1_REG);
+ const root2 = Reg.at(regs.HP_SYS_CLKRST_ROOT_CLK_CTRL2_REG);
+
+ const cpu_div = Field.of(regs.HP_SYS_CLKRST_REG_CPU_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_CPU_CLK_DIV_NUM_V);
+ const cpu_num = Field.of(regs.HP_SYS_CLKRST_REG_CPU_CLK_DIV_NUMERATOR_S, regs.HP_SYS_CLKRST_REG_CPU_CLK_DIV_NUMERATOR_V);
+ const cpu_den = Field.of(regs.HP_SYS_CLKRST_REG_CPU_CLK_DIV_DENOMINATOR_S, regs.HP_SYS_CLKRST_REG_CPU_CLK_DIV_DENOMINATOR_V);
+ const mem_div = Field.of(regs.HP_SYS_CLKRST_REG_MEM_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_MEM_CLK_DIV_NUM_V);
+ const sys_div = Field.of(regs.HP_SYS_CLKRST_REG_SYS_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_SYS_CLK_DIV_NUM_V);
+ const apb_div = Field.of(regs.HP_SYS_CLKRST_REG_APB_CLK_DIV_NUM_S, regs.HP_SYS_CLKRST_REG_APB_CLK_DIV_NUM_V);
+ const update = Field.of(regs.HP_SYS_CLKRST_REG_SOC_CLK_DIV_UPDATE_S, regs.HP_SYS_CLKRST_REG_SOC_CLK_DIV_UPDATE_V);
+
+ // Every divider register holds `divider - 1`.
+ const plan: struct { cpu: u32, mem: u32, sys: u32, apb: u32 } = switch (target) {
+ .mhz360 => .{ .cpu = 1, .mem = 2, .sys = 1, .apb = 2 },
+ .mhz180 => .{ .cpu = 2, .mem = 1, .sys = 1, .apb = 2 },
+ .mhz90 => .{ .cpu = 4, .mem = 1, .sys = 1, .apb = 1 },
+ };
+
+ // The update bit is self-clearing and gates the whole divider set at once. Bounded, because an
+ // unbounded spin on a board with no debugger is indistinguishable from a crash.
+ const commit = struct {
+ fn go(r: Reg, f: Field) void {
+ r.modify(.{f.is(1)});
+ _ = r.waitFor(f, 0, 100_000);
+ }
+ }.go;
+
+ // Upscaling only: this firmware boots at 90 and never lowers. Doing it in the downscale order
+ // would leave APB above its 100 MHz limit while CPU was already fast.
+ root2.modify(.{apb_div.is(plan.apb - 1)});
+ commit(root0, update);
+ root1.modify(.{sys_div.is(plan.sys - 1)});
+ commit(root0, update);
+ root1.modify(.{mem_div.is(plan.mem - 1)});
+ commit(root0, update);
+ root0.modify(.{ cpu_div.is(plan.cpu - 1), cpu_num.is(0), cpu_den.is(0) });
+ commit(root0, update);
+
+ // The source mux is NOT covered by the update bit and must move last; it is already CPLL here,
+ // so this is a no-op that documents the requirement rather than a step that changes anything.
+ //
+ // Then tell the mask ROM, because `ets_delay_us` and anything else built on `g_ticks_per_us`
+ // would otherwise delay by the wrong factor. `ets_update_cpu_frequency` is the recalibrator
+ // (`esp32p4.rom.ld:32`, 0x4fc00044).
+ ets_update_cpu_frequency(@intFromEnum(target));
+}
+
+/// Declared here rather than reached through `soc.rom`, because `soc` imports `hal` and the edge
+/// cannot run both ways. It is a bare linker symbol either way - `build.zig` defines the address
+/// once for the whole image - so a second declaration of it costs nothing and keeps the frequency
+/// change and its recalibration in one function, where forgetting the second is impossible.
+extern fn ets_update_cpu_frequency(mhz: u32) void;
diff --git a/src/pardes/app.zig b/src/pardes/app.zig
index 2f13865..5b1431f 100644
--- a/src/pardes/app.zig
+++ b/src/pardes/app.zig
@@ -185,6 +185,15 @@ export fn zig_main() noreturn {
@as(u32, @intCast(heap.len / 1024)),
});
+ // The CPU clock, before anything is timed against it. The bootloader leaves 90 MHz and the
+ // CPLL is already at 360, so this is a divider change that disturbs neither UART0 (XTAL) nor
+ // the systimer (XTAL/2.5) nor the flash interface (SPLL). See hal/clkrst.zig:setCpuFreq.
+ if (config.cpu_mhz != 90) hal.clkrst.setCpuFreq(switch (config.cpu_mhz) {
+ 180 => .mhz180,
+ 360 => .mhz360,
+ else => .mhz90,
+ });
+
const rwdt_was_armed = hal.rwdt.disable();
hal.systimer.init();
_ = rwdt_was_armed;
@@ -208,6 +217,26 @@ export fn zig_main() noreturn {
});
while (true) {}
}
+
+ // The CPU clock, measured rather than assumed. Every cycle count this firmware reports is
+ // divided by it somewhere, and `src/io/chip.zig` records it as "a measured ~90 MHz" that
+ // nothing here reconfigures - so it is worth printing rather than remembering. The systimer is
+ // XTAL/2.5 = 16 MHz and is NOT derived from the CPU clock (`hal/systimer.zig:31`,
+ // `clk_tree_defs.h:196-198`), which is exactly what makes it a valid reference for measuring it.
+ if (prof) {
+ const t_start = hal.systimer.micros(.unit0) orelse 0;
+ const c_start = soc.cycles();
+ // 50 ms is long enough that the systimer's 16 MHz granularity and the loop's own overhead
+ // are both noise, and short enough to be invisible in a boot.
+ while ((hal.systimer.micros(.unit0) orelse 0) -% t_start < 50_000) {}
+ const elapsed_us = (hal.systimer.micros(.unit0) orelse 0) -% t_start;
+ const elapsed_cy = soc.cycles() - c_start;
+ soc.rom.print("MARK CPU_HZ cycles=%u us=%u khz=%u\r\n", .{
+ @as(u32, @intCast(elapsed_cy)),
+ @as(u32, @intCast(elapsed_us)),
+ @as(u32, @intCast(if (elapsed_us > 0) elapsed_cy * 1000 / elapsed_us else 0)),
+ });
+ }
soc.rom.print("MARK PARDES_READY\r\n", .{});
var in: [256]u8 = undefined;
diff --git a/src/soc.zig b/src/soc.zig
index 350948a..1e94874 100644
--- a/src/soc.zig
+++ b/src/soc.zig
@@ -94,6 +94,11 @@ pub const rom = struct {
pub extern fn ets_printf(fmt: [*:0]const u8, ...) c_int;
pub extern fn ets_delay_us(us: u32) void;
+ /// Tell the ROM the CPU's new frequency, in MHz. `ets_delay_us` and everything else built on
+ /// `g_ticks_per_us` busy-waits by a cycle count derived from it, so a clock change without this
+ /// makes every ROM delay wrong by exactly the ratio. `esp32p4.rom.ld:32`, 0x4fc00044.
+ pub extern fn ets_update_cpu_frequency(mhz: u32) void;
+
/// Invalidate the caches. `map` selects which, from `rom/cache.h:228-236`:
/// L1 ICache0 = 1, ICache1 = 2, L1 DCache = 0x10, L2 = 0x20; `cache_all` is all four.
///