//! Using the ESP32-P4's vendor ISA extensions (xespv2p1 / xesploop1p0, "PIE") from stock Zig. //! //! Two separate gaps hide behind "LLVM does not support the vendor extensions": //! //! 1. The optimiser will never *choose* an `esp.*` instruction, because LLVM has no cost model or //! intrinsics for the PIE unit. That is a performance ceiling: you get no autovectorisation. //! 2. The assembler cannot *encode* the mnemonics. Inline assembly does **not** help - it goes //! through the same integrated assembler: //! //! asm volatile ("esp.vld.128.ip q0, a0, 16") //! -> error: :1:2: unrecognized instruction mnemonic //! //! That is the real blocker, and it is what stops Zig from assembling ESP-IDF's own FreeRTOS //! context switch (portasm.S, 34 instructions under `#if SOC_CPU_HAS_PIE`). //! //! Gap 2 closes without a compiler fork: emit the words with `.insn` and pin the registers the //! fixed encoding names. `tools/encode.sh` generates the words with Espressif's GAS as a //! build-time oracle; every constant below carries the mnemonic it came from. Two of them //! (vld/vst q0) were also read straight out of an ESP-IDF build of this board with Espressif's //! objdump, and agree byte for byte. //! //! Gap 1 does not close this way - hand-written kernels only. But that is what DSP code does //! anyway: esp-dsp is hand-written assembly even under Espressif's own fork. const std = @import("std"); const soc = @import("soc"); pub const panic = std.debug.FullPanic(struct { fn call(msg: []const u8, _: ?usize) noreturn { soc.rom.print("MARK PIE_PANIC %s\r\n", .{msg.ptr}); while (true) {} } }.call); /// CSR 0x7F2 is `CSR_PIE_STATE_REG`; ESP-IDF writes 1 to enable the unit /// (riscv/include/riscv/rv_utils.h:341, riscv/include/riscv/csr_pie.h:19). inline fn enablePie() void { asm volatile ("csrw 0x7f2, 1"); } /// 16 bytes through the PIE register file: one load, one store, neither spellable by the assembler. /// Both instructions post-increment their base register, so a0 is an in-out operand. fn pieCopy16(dst: *align(16) volatile [16]u8, src: *align(16) const volatile [16]u8) void { var p: usize = @intFromPtr(src); asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16 : [p] "={a0}" (p), : [in] "{a0}" (p), : .{ .memory = true }); var q: usize = @intFromPtr(dst); asm volatile (".insn 4, 0x8201223b" // esp.vst.128.ip q0, a0, 16 : [q] "={a0}" (q), : [in] "{a0}" (q), : .{ .memory = true }); } /// The reason the extension exists: 16 signed 8-bit multiply-accumulates in one instruction. /// /// `a` and `b` are 16 bytes each; QACC accumulates 16 lanes which are then spilled as 64 bytes. /// The four `st.qacc` instructions dump the accumulator in quarters, chaining through a0. /// /// q0/q1 stay live across the two asm blocks. That is only sound because nothing else in this /// image emits a PIE instruction - LLVM cannot see the q registers, so it cannot preserve them. fn pieMacS8( out: *align(16) volatile [64]u8, a: *align(16) const volatile [16]i8, b: *align(16) const volatile [16]i8, ) void { var p: usize = @intFromPtr(a); asm volatile ( \\ .insn 4, 0x0201223b # esp.vld.128.ip q0, a0, 16 : [p] "={a0}" (p), : [in] "{a0}" (p), : .{ .memory = true }); p = @intFromPtr(b); asm volatile ( \\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16 : [p] "={a0}" (p), : [in] "{a0}" (p), : .{ .memory = true }); var q: usize = @intFromPtr(out); asm volatile ( \\ .insn 4, 0x0000025b # esp.zero.qacc \\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1 \\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16 \\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16 \\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16 \\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16 : [q] "={a0}" (q), : [in] "{a0}" (q), : .{ .memory = true }); } var source: [16]u8 align(16) = .{ 0x42, 0x00, 0xca, 0xfe, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 }; var dest: [16]u8 align(16) = @splat(0); var vec_a: [16]i8 align(16) = @splat(1); var vec_b: [16]i8 align(16) = .{ 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16 }; var qacc: [64]u8 align(16) = @splat(0); var big_a: [256]i8 align(16) = blk: { var v: [256]i8 = undefined; for (&v, 0..) |*e, i| e.* = @intCast((i % 7) + 1); break :blk v; }; var big_b: [256]i8 align(16) = blk: { var v: [256]i8 = undefined; for (&v, 0..) |*e, i| e.* = @intCast((i % 5) + 1); break :blk v; }; /// Gap 1, measured: a 256-element signed dot product, the way the compiler emits it versus the way /// the PIE unit does it. LLVM cannot autovectorise into `esp.*`, so the scalar loop is what any /// stock-toolchain build gets - with Espressif's fork too, since their LLVM has no autovectoriser /// for this unit either. The vector version is sixteen MACs per instruction with the accumulator /// spilled once at the end. fn dotScalar(a: *const volatile [256]i8, b: *const volatile [256]i8) i32 { var acc: i32 = 0; for (0..256) |i| acc += @as(i32, a[i]) * @as(i32, b[i]); return acc; } fn dotPie(out: *align(16) volatile [64]u8, a: *align(16) const volatile [256]i8, b: *align(16) const volatile [256]i8) i32 { asm volatile (".insn 4, 0x0000025b" ::: .{ .memory = true }); // esp.zero.qacc var pa: usize = @intFromPtr(a); var pb: usize = @intFromPtr(b); for (0..16) |_| { asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16 : [p] "={a0}" (pa), : [in] "{a0}" (pa), : .{ .memory = true }); asm volatile ( \\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16 \\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1 : [p] "={a0}" (pb), : [in] "{a0}" (pb), : .{ .memory = true }); } var q: usize = @intFromPtr(out); asm volatile ( \\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16 \\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16 \\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16 \\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16 : [q] "={a0}" (q), : [in] "{a0}" (q), : .{ .memory = true }); // Sixteen lane accumulators, horizontally summed by the scalar core. var acc: i32 = 0; for (0..16) |i| acc += @as(i32, @bitCast(@as(u32, out[i*4]) | @as(u32, out[i*4+1]) << 8 | @as(u32, out[i*4+2]) << 16 | @as(u32, out[i*4+3]) << 24)); return acc; } export fn zig_main() noreturn { soc.rom.print("\r\nMARK PIE_START csr 0x7f2 <- 1\r\n", .{}); enablePie(); soc.rom.print("MARK PIE_ENABLED survived the CSR write\r\n", .{}); // 1. vector load/store. pieCopy16(&dest, &source); var copy_ok = true; for (source, dest) |x, y| { if (x != y) copy_ok = false; } soc.rom.print("MARK PIE_COPY dest=%02x%02x%02x%02x match=%u\r\n", .{ @as(u32, dest[0]), @as(u32, dest[1]), @as(u32, dest[2]), @as(u32, dest[3]), @as(u32, @intFromBool(copy_ok)), }); // 2. sixteen 8-bit MACs in one instruction. a is all ones and b is 1..16, so the sixteen // accumulator lanes must hold exactly 1..16 - whatever order the QACC dump puts them in. pieMacS8(&qacc, &vec_a, &vec_b); var seen: u32 = 0; var lanes: u32 = 0; for (0..16) |i| { const v = std.mem.readInt(u32, qacc[i * 4 ..][0..4], .little); if (v >= 1 and v <= 16) { seen |= @as(u32, 1) << @intCast(v - 1); lanes += 1; } } soc.rom.print("MARK PIE_MAC lanes=%u distinct=%08x expect=0000ffff sum_ok=%u\r\n", .{ lanes, seen, @as(u32, @intFromBool(seen == 0xffff)), }); // 3. gap 1, in cycles: the same 256-element dot product both ways. Both results must agree, // or the vector path is not computing what the compiler's loop computes. const t0 = soc.cycles(); const scalar = dotScalar(&big_a, &big_b); const t1 = soc.cycles(); const vector = dotPie(&qacc, &big_a, &big_b); const t2 = soc.cycles(); soc.rom.print("MARK PIE_DOT scalar=%d in %u cyc, vector=%d in %u cyc, agree=%u\r\n", .{ scalar, @as(u32, @truncate(t1 - t0)), vector, @as(u32, @truncate(t2 - t1)), @as(u32, @intFromBool(scalar == vector)), }); soc.rom.print("MARK PIE_DONE vendor vector ISA executed from a zig-built image\r\n", .{}); soc.gpio.configureOutput(20); while (true) { soc.gpio.setHigh(20); soc.rom.ets_delay_us(250_000); soc.gpio.setLow(20); soc.rom.ets_delay_us(250_000); } } export fn _start() linksection(".text.entry") callconv(.naked) noreturn { asm volatile ( \\ li t0, 1 << 13 \\ csrs mstatus, t0 \\ la sp, __stack_top \\ la t0, __bss_start \\ la t1, __bss_end \\ bgeu t0, t1, 2f \\1: \\ sw zero, 0(t0) \\ addi t0, t0, 4 \\ bltu t0, t1, 1b \\2: \\ j zig_main ); }