summaryrefslogtreecommitdiff
path: root/examples/pie.zig
diff options
context:
space:
mode:
authorGabriel Schneider <[email protected]>2026-08-25 12:40:53 -0300
committerGabriel Schneider <[email protected]>2026-08-25 12:46:51 -0300
commitf5f8068fac59b4f16046c2022c2fc7c7e447ef4c (patch)
tree2731a3ed4e51cae09e184e25778eded5fc37d1f5 /examples/pie.zig
downloadesp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.tar.gz
esp32p4-f5f8068fac59b4f16046c2022c2fc7c7e447ef4c.zip
zig-p4: pure-Zig ESP32-P4 toolchain
build.zig generates the linker script and drives Zig's own LLD; tools/image.zig turns the ELF into a flashable image and tools/{rom,serial}.zig speak the mask ROM loader over the UART. No CMake, ninja, idf.py, esptool, or external linker. src/soc.zig is a comptime register model over ESP-IDF's own *_reg.h headers; src/hal/ adds peripheral sequences; src/io/ implements std.Io for the chip; src/oracle/ diffs this HAL against ESP-IDF's on the die.
Diffstat (limited to 'examples/pie.zig')
-rw-r--r--examples/pie.zig224
1 files changed, 224 insertions, 0 deletions
diff --git a/examples/pie.zig b/examples/pie.zig
new file mode 100644
index 0000000..9c51cc3
--- /dev/null
+++ b/examples/pie.zig
@@ -0,0 +1,224 @@
+//! Using the ESP32-P4's vendor ISA extensions (xespv2p1 / xesploop1p0, "PIE") from stock Zig.
+//!
+//! Two separate gaps hide behind "LLVM does not support the vendor extensions":
+//!
+//! 1. The optimiser will never *choose* an `esp.*` instruction, because LLVM has no cost model or
+//! intrinsics for the PIE unit. That is a performance ceiling: you get no autovectorisation.
+//! 2. The assembler cannot *encode* the mnemonics. Inline assembly does **not** help - it goes
+//! through the same integrated assembler:
+//!
+//! asm volatile ("esp.vld.128.ip q0, a0, 16")
+//! -> error: <inline asm>:1:2: unrecognized instruction mnemonic
+//!
+//! That is the real blocker, and it is what stops Zig from assembling ESP-IDF's own FreeRTOS
+//! context switch (portasm.S, 34 instructions under `#if SOC_CPU_HAS_PIE`).
+//!
+//! Gap 2 closes without a compiler fork: emit the words with `.insn` and pin the registers the
+//! fixed encoding names. `tools/encode.sh` generates the words with Espressif's GAS as a
+//! build-time oracle; every constant below carries the mnemonic it came from. Two of them
+//! (vld/vst q0) were also read straight out of an ESP-IDF build of this board with Espressif's
+//! objdump, and agree byte for byte.
+//!
+//! Gap 1 does not close this way - hand-written kernels only. But that is what DSP code does
+//! anyway: esp-dsp is hand-written assembly even under Espressif's own fork.
+
+const std = @import("std");
+const soc = @import("soc");
+
+pub const panic = std.debug.FullPanic(struct {
+ fn call(msg: []const u8, _: ?usize) noreturn {
+ soc.rom.print("MARK PIE_PANIC %s\r\n", .{msg.ptr});
+ while (true) {}
+ }
+}.call);
+
+/// CSR 0x7F2 is `CSR_PIE_STATE_REG`; ESP-IDF writes 1 to enable the unit
+/// (riscv/include/riscv/rv_utils.h:341, riscv/include/riscv/csr_pie.h:19).
+inline fn enablePie() void {
+ asm volatile ("csrw 0x7f2, 1");
+}
+
+/// 16 bytes through the PIE register file: one load, one store, neither spellable by the assembler.
+/// Both instructions post-increment their base register, so a0 is an in-out operand.
+fn pieCopy16(dst: *align(16) volatile [16]u8, src: *align(16) const volatile [16]u8) void {
+ var p: usize = @intFromPtr(src);
+ asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16
+ : [p] "={a0}" (p),
+ : [in] "{a0}" (p),
+ : .{ .memory = true });
+ var q: usize = @intFromPtr(dst);
+ asm volatile (".insn 4, 0x8201223b" // esp.vst.128.ip q0, a0, 16
+ : [q] "={a0}" (q),
+ : [in] "{a0}" (q),
+ : .{ .memory = true });
+}
+
+/// The reason the extension exists: 16 signed 8-bit multiply-accumulates in one instruction.
+///
+/// `a` and `b` are 16 bytes each; QACC accumulates 16 lanes which are then spilled as 64 bytes.
+/// The four `st.qacc` instructions dump the accumulator in quarters, chaining through a0.
+///
+/// q0/q1 stay live across the two asm blocks. That is only sound because nothing else in this
+/// image emits a PIE instruction - LLVM cannot see the q registers, so it cannot preserve them.
+fn pieMacS8(
+ out: *align(16) volatile [64]u8,
+ a: *align(16) const volatile [16]i8,
+ b: *align(16) const volatile [16]i8,
+) void {
+ var p: usize = @intFromPtr(a);
+ asm volatile (
+ \\ .insn 4, 0x0201223b # esp.vld.128.ip q0, a0, 16
+ : [p] "={a0}" (p),
+ : [in] "{a0}" (p),
+ : .{ .memory = true });
+ p = @intFromPtr(b);
+ asm volatile (
+ \\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16
+ : [p] "={a0}" (p),
+ : [in] "{a0}" (p),
+ : .{ .memory = true });
+
+ var q: usize = @intFromPtr(out);
+ asm volatile (
+ \\ .insn 4, 0x0000025b # esp.zero.qacc
+ \\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1
+ \\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16
+ \\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16
+ \\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16
+ \\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16
+ : [q] "={a0}" (q),
+ : [in] "{a0}" (q),
+ : .{ .memory = true });
+}
+
+var source: [16]u8 align(16) = .{ 0x42, 0x00, 0xca, 0xfe, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 };
+var dest: [16]u8 align(16) = @splat(0);
+
+var vec_a: [16]i8 align(16) = @splat(1);
+var vec_b: [16]i8 align(16) = .{ 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16 };
+var qacc: [64]u8 align(16) = @splat(0);
+
+var big_a: [256]i8 align(16) = blk: {
+ var v: [256]i8 = undefined;
+ for (&v, 0..) |*e, i| e.* = @intCast((i % 7) + 1);
+ break :blk v;
+};
+var big_b: [256]i8 align(16) = blk: {
+ var v: [256]i8 = undefined;
+ for (&v, 0..) |*e, i| e.* = @intCast((i % 5) + 1);
+ break :blk v;
+};
+
+/// Gap 1, measured: a 256-element signed dot product, the way the compiler emits it versus the way
+/// the PIE unit does it. LLVM cannot autovectorise into `esp.*`, so the scalar loop is what any
+/// stock-toolchain build gets - with Espressif's fork too, since their LLVM has no autovectoriser
+/// for this unit either. The vector version is sixteen MACs per instruction with the accumulator
+/// spilled once at the end.
+fn dotScalar(a: *const volatile [256]i8, b: *const volatile [256]i8) i32 {
+ var acc: i32 = 0;
+ for (0..256) |i| acc += @as(i32, a[i]) * @as(i32, b[i]);
+ return acc;
+}
+
+fn dotPie(out: *align(16) volatile [64]u8, a: *align(16) const volatile [256]i8, b: *align(16) const volatile [256]i8) i32 {
+ asm volatile (".insn 4, 0x0000025b" ::: .{ .memory = true }); // esp.zero.qacc
+ var pa: usize = @intFromPtr(a);
+ var pb: usize = @intFromPtr(b);
+ for (0..16) |_| {
+ asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16
+ : [p] "={a0}" (pa),
+ : [in] "{a0}" (pa),
+ : .{ .memory = true });
+ asm volatile (
+ \\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16
+ \\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1
+ : [p] "={a0}" (pb),
+ : [in] "{a0}" (pb),
+ : .{ .memory = true });
+ }
+ var q: usize = @intFromPtr(out);
+ asm volatile (
+ \\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16
+ \\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16
+ \\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16
+ \\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16
+ : [q] "={a0}" (q),
+ : [in] "{a0}" (q),
+ : .{ .memory = true });
+ // Sixteen lane accumulators, horizontally summed by the scalar core.
+ var acc: i32 = 0;
+ for (0..16) |i| acc += @as(i32, @bitCast(@as(u32, out[i*4]) | @as(u32, out[i*4+1]) << 8 | @as(u32, out[i*4+2]) << 16 | @as(u32, out[i*4+3]) << 24));
+ return acc;
+}
+
+export fn zig_main() noreturn {
+ soc.rom.print("\r\nMARK PIE_START csr 0x7f2 <- 1\r\n", .{});
+ enablePie();
+ soc.rom.print("MARK PIE_ENABLED survived the CSR write\r\n", .{});
+
+ // 1. vector load/store.
+ pieCopy16(&dest, &source);
+ var copy_ok = true;
+ for (source, dest) |x, y| {
+ if (x != y) copy_ok = false;
+ }
+ soc.rom.print("MARK PIE_COPY dest=%02x%02x%02x%02x match=%u\r\n", .{
+ @as(u32, dest[0]), @as(u32, dest[1]), @as(u32, dest[2]), @as(u32, dest[3]),
+ @as(u32, @intFromBool(copy_ok)),
+ });
+
+ // 2. sixteen 8-bit MACs in one instruction. a is all ones and b is 1..16, so the sixteen
+ // accumulator lanes must hold exactly 1..16 - whatever order the QACC dump puts them in.
+ pieMacS8(&qacc, &vec_a, &vec_b);
+ var seen: u32 = 0;
+ var lanes: u32 = 0;
+ for (0..16) |i| {
+ const v = std.mem.readInt(u32, qacc[i * 4 ..][0..4], .little);
+ if (v >= 1 and v <= 16) {
+ seen |= @as(u32, 1) << @intCast(v - 1);
+ lanes += 1;
+ }
+ }
+ soc.rom.print("MARK PIE_MAC lanes=%u distinct=%08x expect=0000ffff sum_ok=%u\r\n", .{
+ lanes, seen, @as(u32, @intFromBool(seen == 0xffff)),
+ });
+
+ // 3. gap 1, in cycles: the same 256-element dot product both ways. Both results must agree,
+ // or the vector path is not computing what the compiler's loop computes.
+ const t0 = soc.cycles();
+ const scalar = dotScalar(&big_a, &big_b);
+ const t1 = soc.cycles();
+ const vector = dotPie(&qacc, &big_a, &big_b);
+ const t2 = soc.cycles();
+ soc.rom.print("MARK PIE_DOT scalar=%d in %u cyc, vector=%d in %u cyc, agree=%u\r\n", .{
+ scalar, @as(u32, @truncate(t1 - t0)),
+ vector, @as(u32, @truncate(t2 - t1)),
+ @as(u32, @intFromBool(scalar == vector)),
+ });
+ soc.rom.print("MARK PIE_DONE vendor vector ISA executed from a zig-built image\r\n", .{});
+
+ soc.gpio.configureOutput(20);
+ while (true) {
+ soc.gpio.setHigh(20);
+ soc.rom.ets_delay_us(250_000);
+ soc.gpio.setLow(20);
+ soc.rom.ets_delay_us(250_000);
+ }
+}
+
+export fn _start() linksection(".text.entry") callconv(.naked) noreturn {
+ asm volatile (
+ \\ li t0, 1 << 13
+ \\ csrs mstatus, t0
+ \\ la sp, __stack_top
+ \\ la t0, __bss_start
+ \\ la t1, __bss_end
+ \\ bgeu t0, t1, 2f
+ \\1:
+ \\ sw zero, 0(t0)
+ \\ addi t0, t0, 4
+ \\ bltu t0, t1, 1b
+ \\2:
+ \\ j zig_main
+ );
+}