diff options
Diffstat (limited to 'examples/pie.zig')
| -rw-r--r-- | examples/pie.zig | 224 |
1 files changed, 224 insertions, 0 deletions
diff --git a/examples/pie.zig b/examples/pie.zig new file mode 100644 index 0000000..9c51cc3 --- /dev/null +++ b/examples/pie.zig @@ -0,0 +1,224 @@ +//! Using the ESP32-P4's vendor ISA extensions (xespv2p1 / xesploop1p0, "PIE") from stock Zig. +//! +//! Two separate gaps hide behind "LLVM does not support the vendor extensions": +//! +//! 1. The optimiser will never *choose* an `esp.*` instruction, because LLVM has no cost model or +//! intrinsics for the PIE unit. That is a performance ceiling: you get no autovectorisation. +//! 2. The assembler cannot *encode* the mnemonics. Inline assembly does **not** help - it goes +//! through the same integrated assembler: +//! +//! asm volatile ("esp.vld.128.ip q0, a0, 16") +//! -> error: <inline asm>:1:2: unrecognized instruction mnemonic +//! +//! That is the real blocker, and it is what stops Zig from assembling ESP-IDF's own FreeRTOS +//! context switch (portasm.S, 34 instructions under `#if SOC_CPU_HAS_PIE`). +//! +//! Gap 2 closes without a compiler fork: emit the words with `.insn` and pin the registers the +//! fixed encoding names. `tools/encode.sh` generates the words with Espressif's GAS as a +//! build-time oracle; every constant below carries the mnemonic it came from. Two of them +//! (vld/vst q0) were also read straight out of an ESP-IDF build of this board with Espressif's +//! objdump, and agree byte for byte. +//! +//! Gap 1 does not close this way - hand-written kernels only. But that is what DSP code does +//! anyway: esp-dsp is hand-written assembly even under Espressif's own fork. + +const std = @import("std"); +const soc = @import("soc"); + +pub const panic = std.debug.FullPanic(struct { + fn call(msg: []const u8, _: ?usize) noreturn { + soc.rom.print("MARK PIE_PANIC %s\r\n", .{msg.ptr}); + while (true) {} + } +}.call); + +/// CSR 0x7F2 is `CSR_PIE_STATE_REG`; ESP-IDF writes 1 to enable the unit +/// (riscv/include/riscv/rv_utils.h:341, riscv/include/riscv/csr_pie.h:19). +inline fn enablePie() void { + asm volatile ("csrw 0x7f2, 1"); +} + +/// 16 bytes through the PIE register file: one load, one store, neither spellable by the assembler. +/// Both instructions post-increment their base register, so a0 is an in-out operand. +fn pieCopy16(dst: *align(16) volatile [16]u8, src: *align(16) const volatile [16]u8) void { + var p: usize = @intFromPtr(src); + asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16 + : [p] "={a0}" (p), + : [in] "{a0}" (p), + : .{ .memory = true }); + var q: usize = @intFromPtr(dst); + asm volatile (".insn 4, 0x8201223b" // esp.vst.128.ip q0, a0, 16 + : [q] "={a0}" (q), + : [in] "{a0}" (q), + : .{ .memory = true }); +} + +/// The reason the extension exists: 16 signed 8-bit multiply-accumulates in one instruction. +/// +/// `a` and `b` are 16 bytes each; QACC accumulates 16 lanes which are then spilled as 64 bytes. +/// The four `st.qacc` instructions dump the accumulator in quarters, chaining through a0. +/// +/// q0/q1 stay live across the two asm blocks. That is only sound because nothing else in this +/// image emits a PIE instruction - LLVM cannot see the q registers, so it cannot preserve them. +fn pieMacS8( + out: *align(16) volatile [64]u8, + a: *align(16) const volatile [16]i8, + b: *align(16) const volatile [16]i8, +) void { + var p: usize = @intFromPtr(a); + asm volatile ( + \\ .insn 4, 0x0201223b # esp.vld.128.ip q0, a0, 16 + : [p] "={a0}" (p), + : [in] "{a0}" (p), + : .{ .memory = true }); + p = @intFromPtr(b); + asm volatile ( + \\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16 + : [p] "={a0}" (p), + : [in] "{a0}" (p), + : .{ .memory = true }); + + var q: usize = @intFromPtr(out); + asm volatile ( + \\ .insn 4, 0x0000025b # esp.zero.qacc + \\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1 + \\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16 + \\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16 + \\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16 + \\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16 + : [q] "={a0}" (q), + : [in] "{a0}" (q), + : .{ .memory = true }); +} + +var source: [16]u8 align(16) = .{ 0x42, 0x00, 0xca, 0xfe, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 }; +var dest: [16]u8 align(16) = @splat(0); + +var vec_a: [16]i8 align(16) = @splat(1); +var vec_b: [16]i8 align(16) = .{ 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16 }; +var qacc: [64]u8 align(16) = @splat(0); + +var big_a: [256]i8 align(16) = blk: { + var v: [256]i8 = undefined; + for (&v, 0..) |*e, i| e.* = @intCast((i % 7) + 1); + break :blk v; +}; +var big_b: [256]i8 align(16) = blk: { + var v: [256]i8 = undefined; + for (&v, 0..) |*e, i| e.* = @intCast((i % 5) + 1); + break :blk v; +}; + +/// Gap 1, measured: a 256-element signed dot product, the way the compiler emits it versus the way +/// the PIE unit does it. LLVM cannot autovectorise into `esp.*`, so the scalar loop is what any +/// stock-toolchain build gets - with Espressif's fork too, since their LLVM has no autovectoriser +/// for this unit either. The vector version is sixteen MACs per instruction with the accumulator +/// spilled once at the end. +fn dotScalar(a: *const volatile [256]i8, b: *const volatile [256]i8) i32 { + var acc: i32 = 0; + for (0..256) |i| acc += @as(i32, a[i]) * @as(i32, b[i]); + return acc; +} + +fn dotPie(out: *align(16) volatile [64]u8, a: *align(16) const volatile [256]i8, b: *align(16) const volatile [256]i8) i32 { + asm volatile (".insn 4, 0x0000025b" ::: .{ .memory = true }); // esp.zero.qacc + var pa: usize = @intFromPtr(a); + var pb: usize = @intFromPtr(b); + for (0..16) |_| { + asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16 + : [p] "={a0}" (pa), + : [in] "{a0}" (pa), + : .{ .memory = true }); + asm volatile ( + \\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16 + \\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1 + : [p] "={a0}" (pb), + : [in] "{a0}" (pb), + : .{ .memory = true }); + } + var q: usize = @intFromPtr(out); + asm volatile ( + \\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16 + \\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16 + \\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16 + \\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16 + : [q] "={a0}" (q), + : [in] "{a0}" (q), + : .{ .memory = true }); + // Sixteen lane accumulators, horizontally summed by the scalar core. + var acc: i32 = 0; + for (0..16) |i| acc += @as(i32, @bitCast(@as(u32, out[i*4]) | @as(u32, out[i*4+1]) << 8 | @as(u32, out[i*4+2]) << 16 | @as(u32, out[i*4+3]) << 24)); + return acc; +} + +export fn zig_main() noreturn { + soc.rom.print("\r\nMARK PIE_START csr 0x7f2 <- 1\r\n", .{}); + enablePie(); + soc.rom.print("MARK PIE_ENABLED survived the CSR write\r\n", .{}); + + // 1. vector load/store. + pieCopy16(&dest, &source); + var copy_ok = true; + for (source, dest) |x, y| { + if (x != y) copy_ok = false; + } + soc.rom.print("MARK PIE_COPY dest=%02x%02x%02x%02x match=%u\r\n", .{ + @as(u32, dest[0]), @as(u32, dest[1]), @as(u32, dest[2]), @as(u32, dest[3]), + @as(u32, @intFromBool(copy_ok)), + }); + + // 2. sixteen 8-bit MACs in one instruction. a is all ones and b is 1..16, so the sixteen + // accumulator lanes must hold exactly 1..16 - whatever order the QACC dump puts them in. + pieMacS8(&qacc, &vec_a, &vec_b); + var seen: u32 = 0; + var lanes: u32 = 0; + for (0..16) |i| { + const v = std.mem.readInt(u32, qacc[i * 4 ..][0..4], .little); + if (v >= 1 and v <= 16) { + seen |= @as(u32, 1) << @intCast(v - 1); + lanes += 1; + } + } + soc.rom.print("MARK PIE_MAC lanes=%u distinct=%08x expect=0000ffff sum_ok=%u\r\n", .{ + lanes, seen, @as(u32, @intFromBool(seen == 0xffff)), + }); + + // 3. gap 1, in cycles: the same 256-element dot product both ways. Both results must agree, + // or the vector path is not computing what the compiler's loop computes. + const t0 = soc.cycles(); + const scalar = dotScalar(&big_a, &big_b); + const t1 = soc.cycles(); + const vector = dotPie(&qacc, &big_a, &big_b); + const t2 = soc.cycles(); + soc.rom.print("MARK PIE_DOT scalar=%d in %u cyc, vector=%d in %u cyc, agree=%u\r\n", .{ + scalar, @as(u32, @truncate(t1 - t0)), + vector, @as(u32, @truncate(t2 - t1)), + @as(u32, @intFromBool(scalar == vector)), + }); + soc.rom.print("MARK PIE_DONE vendor vector ISA executed from a zig-built image\r\n", .{}); + + soc.gpio.configureOutput(20); + while (true) { + soc.gpio.setHigh(20); + soc.rom.ets_delay_us(250_000); + soc.gpio.setLow(20); + soc.rom.ets_delay_us(250_000); + } +} + +export fn _start() linksection(".text.entry") callconv(.naked) noreturn { + asm volatile ( + \\ li t0, 1 << 13 + \\ csrs mstatus, t0 + \\ la sp, __stack_top + \\ la t0, __bss_start + \\ la t1, __bss_end + \\ bgeu t0, t1, 2f + \\1: + \\ sw zero, 0(t0) + \\ addi t0, t0, 4 + \\ bltu t0, t1, 1b + \\2: + \\ j zig_main + ); +} |
