1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
|
//! ESP32-P4 peripherals, modelled at comptime.
//!
//! There is no HAL here and no generated 20k-line register header: a `Reg` is a typed pointer to
//! an MMIO word, and a peripheral is a struct of them. Everything is `inline`, so `gpio.setHigh(20)`
//! compiles to the single `sw` instruction it should be, and a wrong bit index is a compile error
//! rather than a silent write.
//!
//! Addresses are from ESP-IDF v6.0.2 `components/soc/esp32p4/register/hw_ver1/soc/` - the pre-v3
//! header set, which is the one that matches this silicon (rev v1.3) - except the GPIO matrix
//! signal index, which lives in `components/soc/esp32p4/include/soc/gpio_sig_map.h`.
const std = @import("std");
/// A 32-bit memory-mapped register.
pub fn Reg(comptime addr: usize) type {
return struct {
pub const address = addr;
const ptr: *volatile u32 = @ptrFromInt(addr);
pub inline fn read() u32 {
return ptr.*;
}
pub inline fn write(value: u32) void {
ptr.* = value;
}
pub inline fn set(mask: u32) void {
ptr.* = ptr.* | mask;
}
pub inline fn clear(mask: u32) void {
ptr.* = ptr.* & ~mask;
}
/// Read-modify-write a bitfield: `modify(.{ .shift = 12, .width = 3 }, 5)`.
pub inline fn modify(comptime field: Field, value: u32) void {
const mask: u32 = ((@as(u32, 1) << field.width) - 1) << field.shift;
ptr.* = (ptr.* & ~mask) | ((value << field.shift) & mask);
}
};
}
pub const Field = struct { shift: u5, width: u5 };
/// A register with a named layout: pass a packed struct whose bit width is 32 and the accessors
/// become typed, so a pad is configured by naming fields instead of shifting bits. Read-modify-
/// write stays explicit - `var v = reg.read(); v.mcu_sel = 1; reg.write(v);` - because that is one
/// load and one store, and hiding it behind a partial-update type buys nothing here.
pub fn Typed(comptime T: type, comptime addr: usize) type {
comptime std.debug.assert(@bitSizeOf(T) == 32);
return struct {
pub const address = addr;
const ptr: *volatile T = @ptrFromInt(addr);
pub inline fn read() T {
return ptr.*;
}
pub inline fn write(value: T) void {
ptr.* = value;
}
/// Apply `f` to the current value and write the result back.
pub inline fn modify(comptime f: fn (T) T) void {
ptr.* = f(ptr.*);
}
};
}
/// An array of identical registers. The index type is narrowed to the array's real range, so an
/// out-of-range access is a compile error in every optimize mode - an `assert` would have been
/// compiled out under ReleaseSmall, which is this project's default.
pub fn RegArray(comptime base: usize, comptime stride: usize, comptime count: usize) type {
return struct {
pub const Index = std.math.IntFittingRange(0, count - 1);
/// comptime, because `IntFittingRange` rounds up to a whole width: for a 57-entry array the
/// index type is u6, which would happily accept 57..63. Every caller here passes a comptime
/// pin anyway, so this costs nothing and makes the bound real in all optimize modes.
pub inline fn at(comptime index: Index) *volatile u32 {
comptime std.debug.assert(index < count);
return @ptrFromInt(base + @as(usize, index) * stride);
}
};
}
const hp_periph1 = 0x500C0000;
/// GPIO and the IO MUX now live in the HAL, which builds them out of ESP-IDF's own register macros
/// (`hal/gpio.zig`) instead of the hand-transcribed addresses that used to be here. The
/// transcription is exactly the kind of thing that goes quietly wrong: this file's matrix constant
/// said 256 with a comment warning that the S3's is 128, and the first hand-written replacement in
/// the HAL used 128 anyway. It now comes from `SIG_GPIO_OUT_IDX` in IDF's `gpio_sig_map.h`.
pub const gpio = @import("hal").gpio;
/// Mask ROM routines. These are the only "library" a bare image links against: the addresses come
/// from `components/esp_rom/esp32p4/ld/esp32p4.rom.ld` and the linker script re-declares them.
pub const rom = struct {
pub extern fn ets_printf(fmt: [*:0]const u8, ...) c_int;
pub extern fn ets_delay_us(us: u32) void;
/// Tell the ROM the CPU's new frequency, in MHz. `ets_delay_us` and everything else built on
/// `g_ticks_per_us` busy-waits by a cycle count derived from it, so a clock change without this
/// makes every ROM delay wrong by exactly the ratio. `esp32p4.rom.ld:32`, 0x4fc00044.
pub extern fn ets_update_cpu_frequency(mhz: u32) void;
/// Invalidate the caches. `map` selects which, from `rom/cache.h:228-236`:
/// L1 ICache0 = 1, ICache1 = 2, L1 DCache = 0x10, L2 = 0x20; `cache_all` is all four.
///
/// This is not optional on a large image, and finding that out took a while. The second-stage
/// bootloader leaves cache lines behind that do not match the mapping it finally installs, so an
/// application can read its own `.rodata` and get someone else's bytes - deterministically,
/// which is what makes it look like anything other than a cache. Measured: a load at 0x40035a1c
/// returned `93 85 85 0f`, and after touching 512 KiB to evict the line the same load returned
/// `3c ee 08 40`, which is what the image holds there. `_start` calls this before it clears
/// `.bss`, so nothing of ours is in flight when the lines are dropped.
pub extern fn Cache_Invalidate_All(map: u32) c_int;
/// Every cache this chip has: L1 ICache0 | L1 ICache1 | L1 DCache | L2.
pub const cache_all: u32 = 0x33;
pub inline fn print(comptime fmt: [*:0]const u8, args: anytype) void {
_ = @call(.auto, ets_printf, .{fmt} ++ args);
}
};
/// Evict every flash-mapped cache line the bootloader left behind, by reading more flash than the
/// caches can hold. Call it before the first byte of `.rodata` is touched.
///
/// This is a workaround for a real defect in the hand-over, not a tidiness measure. The second-stage
/// bootloader leaves lines cached against a mapping it then replaces, so an application reads its
/// own `.rodata` and gets its own `.text` back - **deterministically**, which is exactly what makes
/// it look like anything other than a cache. Measured on this die: a load at `0x40035A1C` returned
/// `93 85 85 0f` before eviction and `3c ee 08 40` after, and the second is what the image holds
/// there. A string literal read before this runs is machine code, so a firmware whose first act is
/// to print a marker prints garbage and looks like it never booted at all.
///
/// Everything cheaper was tried first and every one of them said the hardware was fine, which is
/// why the list is here rather than being rediscovered:
///
/// * The MMU table is correct. Entries 0..9 read `0x1001`..`0x100a` - the valid bit plus physical
/// page N+1 - which is exactly what the image builder's single flash-to-vaddr anchor requires,
/// and 10..11 are unmapped as they should be. Read off the die through
/// `SPI_MEM_C_MMU_ITEM_INDEX_REG`, not inferred.
/// * The flash is correct. `zig build flash` verifies an MD5 of what the ROM stored, and the
/// image matches the ELF byte for byte at the addresses that misread.
/// * The page size is not in question: hardwired to 64 KiB on this chip
/// (`hal/esp32p4/mmu_ll.h:126-130` returns `MMU_PAGE_64KB` and the setter asserts it).
/// * Not fragmentation of the mapping either: the bad bytes arrive in one contiguous run of
/// >= 192 B, not in 64-byte lines, and they are identical across three resets and two
/// reflashes - determinism is what kept this looking like anything but a cache.
///
/// 512 KiB is four times the 128 KiB the L2 measured at (`examples/memprobe.zig` found real RAM
/// stopping at `0x4FFA0000`, the cache taking the rest), with the L1s smaller still.
///
/// `rom.Cache_Invalidate_All` is the instrument that ought to do this and does not: called from an
/// image the ROM did not launch, it faults inside the ROM with its argument stranded in `a2`.
/// Capacity eviction needs no preconditions, which is the whole reason it is what ships. 512 KiB at
/// a 64-byte stride is 8,192 loads, once per boot.
pub fn flushFlashCache() void {
var sink: u32 = 0;
var p: u32 = 0x4000_0000;
while (p < 0x4008_0000) : (p += 64) {
sink +%= @as(*volatile u32, @ptrFromInt(p)).*;
}
// Consumed through a volatile store, or the optimiser drops the whole loop as dead.
@as(*volatile u32, &flush_sink).* = sink;
}
var flush_sink: u32 = 0;
/// Busy-wait for a number of CPU cycles, using the cycle counter rather than the mask ROM. Useful
/// when an image must not depend on ROM entry points at all, and for delays shorter than the ROM's
/// microsecond granularity.
pub inline fn delayCycles(n: u64) void {
const start = cycles();
while (cycles() - start < n) {}
}
/// Cycle counter: CSR 0xC00/0xC80, i.e. `cycle`/`cycleh` - the unprivileged shadows of mcycle, and
/// what ESP-IDF itself reads on this part (`rv_utils.h`: `RV_READ_CSR(cycle)`, because
/// SOC_CPU_HAS_CSR_PC is not defined for the P4).
///
/// Read high-low-high: two separate CSR reads can straddle a wrap of the low word, which would
/// otherwise report a value 2^32 too large roughly every 47 seconds at 90 MHz.
pub inline fn cycles() u64 {
while (true) {
var hi0: u32 = undefined;
var lo: u32 = undefined;
var hi1: u32 = undefined;
asm volatile ("csrr %[r], 0xC80"
: [r] "=r" (hi0),
);
asm volatile ("csrr %[r], 0xC00"
: [r] "=r" (lo),
);
asm volatile ("csrr %[r], 0xC80"
: [r] "=r" (hi1),
);
if (hi0 == hi1) return (@as(u64, hi0) << 32) | lo;
}
}
|