1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
|
//! Using the ESP32-P4's vendor ISA extensions (xespv2p1 / xesploop1p0, "PIE") from stock Zig.
//!
//! Two separate gaps hide behind "LLVM does not support the vendor extensions":
//!
//! 1. The optimiser will never *choose* an `esp.*` instruction, because LLVM has no cost model or
//! intrinsics for the PIE unit. That is a performance ceiling: you get no autovectorisation.
//! 2. The assembler cannot *encode* the mnemonics. Inline assembly does **not** help - it goes
//! through the same integrated assembler:
//!
//! asm volatile ("esp.vld.128.ip q0, a0, 16")
//! -> error: <inline asm>:1:2: unrecognized instruction mnemonic
//!
//! That is the real blocker, and it is what stops Zig from assembling ESP-IDF's own FreeRTOS
//! context switch (portasm.S, 34 instructions under `#if SOC_CPU_HAS_PIE`).
//!
//! Gap 2 closes without a compiler fork: emit the words with `.insn` and pin the registers the
//! fixed encoding names. `tools/encode.sh` generates the words with Espressif's GAS as a
//! build-time oracle; every constant below carries the mnemonic it came from. Two of them
//! (vld/vst q0) were also read straight out of an ESP-IDF build of this board with Espressif's
//! objdump, and agree byte for byte.
//!
//! Gap 1 does not close this way - hand-written kernels only. But that is what DSP code does
//! anyway: esp-dsp is hand-written assembly even under Espressif's own fork.
const std = @import("std");
const soc = @import("soc");
pub const panic = std.debug.FullPanic(struct {
fn call(msg: []const u8, _: ?usize) noreturn {
soc.rom.print("MARK PIE_PANIC %s\r\n", .{msg.ptr});
while (true) {}
}
}.call);
/// CSR 0x7F2 is `CSR_PIE_STATE_REG`; ESP-IDF writes 1 to enable the unit
/// (riscv/include/riscv/rv_utils.h:341, riscv/include/riscv/csr_pie.h:19).
inline fn enablePie() void {
asm volatile ("csrw 0x7f2, 1");
}
/// 16 bytes through the PIE register file: one load, one store, neither spellable by the assembler.
/// Both instructions post-increment their base register, so a0 is an in-out operand.
fn pieCopy16(dst: *align(16) volatile [16]u8, src: *align(16) const volatile [16]u8) void {
var p: usize = @intFromPtr(src);
asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16
: [p] "={a0}" (p),
: [in] "{a0}" (p),
: .{ .memory = true });
var q: usize = @intFromPtr(dst);
asm volatile (".insn 4, 0x8201223b" // esp.vst.128.ip q0, a0, 16
: [q] "={a0}" (q),
: [in] "{a0}" (q),
: .{ .memory = true });
}
/// The reason the extension exists: 16 signed 8-bit multiply-accumulates in one instruction.
///
/// `a` and `b` are 16 bytes each; QACC accumulates 16 lanes which are then spilled as 64 bytes.
/// The four `st.qacc` instructions dump the accumulator in quarters, chaining through a0.
///
/// q0/q1 stay live across the two asm blocks. That is only sound because nothing else in this
/// image emits a PIE instruction - LLVM cannot see the q registers, so it cannot preserve them.
fn pieMacS8(
out: *align(16) volatile [64]u8,
a: *align(16) const volatile [16]i8,
b: *align(16) const volatile [16]i8,
) void {
var p: usize = @intFromPtr(a);
asm volatile (
\\ .insn 4, 0x0201223b # esp.vld.128.ip q0, a0, 16
: [p] "={a0}" (p),
: [in] "{a0}" (p),
: .{ .memory = true });
p = @intFromPtr(b);
asm volatile (
\\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16
: [p] "={a0}" (p),
: [in] "{a0}" (p),
: .{ .memory = true });
var q: usize = @intFromPtr(out);
asm volatile (
\\ .insn 4, 0x0000025b # esp.zero.qacc
\\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1
\\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16
\\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16
\\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16
\\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16
: [q] "={a0}" (q),
: [in] "{a0}" (q),
: .{ .memory = true });
}
var source: [16]u8 align(16) = .{ 0x42, 0x00, 0xca, 0xfe, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12 };
var dest: [16]u8 align(16) = @splat(0);
var vec_a: [16]i8 align(16) = @splat(1);
var vec_b: [16]i8 align(16) = .{ 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16 };
var qacc: [64]u8 align(16) = @splat(0);
var big_a: [256]i8 align(16) = blk: {
var v: [256]i8 = undefined;
for (&v, 0..) |*e, i| e.* = @intCast((i % 7) + 1);
break :blk v;
};
var big_b: [256]i8 align(16) = blk: {
var v: [256]i8 = undefined;
for (&v, 0..) |*e, i| e.* = @intCast((i % 5) + 1);
break :blk v;
};
/// Gap 1, measured: a 256-element signed dot product, the way the compiler emits it versus the way
/// the PIE unit does it. LLVM cannot autovectorise into `esp.*`, so the scalar loop is what any
/// stock-toolchain build gets - with Espressif's fork too, since their LLVM has no autovectoriser
/// for this unit either. The vector version is sixteen MACs per instruction with the accumulator
/// spilled once at the end.
fn dotScalar(a: *const volatile [256]i8, b: *const volatile [256]i8) i32 {
var acc: i32 = 0;
for (0..256) |i| acc += @as(i32, a[i]) * @as(i32, b[i]);
return acc;
}
fn dotPie(out: *align(16) volatile [64]u8, a: *align(16) const volatile [256]i8, b: *align(16) const volatile [256]i8) i32 {
asm volatile (".insn 4, 0x0000025b" ::: .{ .memory = true }); // esp.zero.qacc
var pa: usize = @intFromPtr(a);
var pb: usize = @intFromPtr(b);
for (0..16) |_| {
asm volatile (".insn 4, 0x0201223b" // esp.vld.128.ip q0, a0, 16
: [p] "={a0}" (pa),
: [in] "{a0}" (pa),
: .{ .memory = true });
asm volatile (
\\ .insn 4, 0x0201263b # esp.vld.128.ip q1, a0, 16
\\ .insn 4, 0x06c5005f # esp.vmulas.s8.qacc q0, q1
: [p] "={a0}" (pb),
: [in] "{a0}" (pb),
: .{ .memory = true });
}
var q: usize = @intFromPtr(out);
asm volatile (
\\ .insn 4, 0xa001423b # esp.st.qacc.l.l.128.ip a0, 16
\\ .insn 4, 0x8001423b # esp.st.qacc.l.h.128.ip a0, 16
\\ .insn 4, 0xe001423b # esp.st.qacc.h.l.128.ip a0, 16
\\ .insn 4, 0xc001423b # esp.st.qacc.h.h.128.ip a0, 16
: [q] "={a0}" (q),
: [in] "{a0}" (q),
: .{ .memory = true });
// Sixteen lane accumulators, horizontally summed by the scalar core.
var acc: i32 = 0;
for (0..16) |i| acc += @as(i32, @bitCast(@as(u32, out[i*4]) | @as(u32, out[i*4+1]) << 8 | @as(u32, out[i*4+2]) << 16 | @as(u32, out[i*4+3]) << 24));
return acc;
}
export fn zig_main() noreturn {
soc.rom.print("\r\nMARK PIE_START csr 0x7f2 <- 1\r\n", .{});
enablePie();
soc.rom.print("MARK PIE_ENABLED survived the CSR write\r\n", .{});
// 1. vector load/store.
pieCopy16(&dest, &source);
var copy_ok = true;
for (source, dest) |x, y| {
if (x != y) copy_ok = false;
}
soc.rom.print("MARK PIE_COPY dest=%02x%02x%02x%02x match=%u\r\n", .{
@as(u32, dest[0]), @as(u32, dest[1]), @as(u32, dest[2]), @as(u32, dest[3]),
@as(u32, @intFromBool(copy_ok)),
});
// 2. sixteen 8-bit MACs in one instruction. a is all ones and b is 1..16, so the sixteen
// accumulator lanes must hold exactly 1..16 - whatever order the QACC dump puts them in.
pieMacS8(&qacc, &vec_a, &vec_b);
var seen: u32 = 0;
var lanes: u32 = 0;
for (0..16) |i| {
const v = std.mem.readInt(u32, qacc[i * 4 ..][0..4], .little);
if (v >= 1 and v <= 16) {
seen |= @as(u32, 1) << @intCast(v - 1);
lanes += 1;
}
}
soc.rom.print("MARK PIE_MAC lanes=%u distinct=%08x expect=0000ffff sum_ok=%u\r\n", .{
lanes, seen, @as(u32, @intFromBool(seen == 0xffff)),
});
// 3. gap 1, in cycles: the same 256-element dot product both ways. Both results must agree,
// or the vector path is not computing what the compiler's loop computes.
const t0 = soc.cycles();
const scalar = dotScalar(&big_a, &big_b);
const t1 = soc.cycles();
const vector = dotPie(&qacc, &big_a, &big_b);
const t2 = soc.cycles();
soc.rom.print("MARK PIE_DOT scalar=%d in %u cyc, vector=%d in %u cyc, agree=%u\r\n", .{
scalar, @as(u32, @truncate(t1 - t0)),
vector, @as(u32, @truncate(t2 - t1)),
@as(u32, @intFromBool(scalar == vector)),
});
soc.rom.print("MARK PIE_DONE vendor vector ISA executed from a zig-built image\r\n", .{});
soc.gpio.configureOutput(20);
while (true) {
soc.gpio.setHigh(20);
soc.rom.ets_delay_us(250_000);
soc.gpio.setLow(20);
soc.rom.ets_delay_us(250_000);
}
}
export fn _start() linksection(".text.entry") callconv(.naked) noreturn {
asm volatile (
\\ li t0, 1 << 13
\\ csrs mstatus, t0
\\ la sp, __stack_top
\\ la t0, __bss_start
\\ la t1, __bss_end
\\ bgeu t0, t1, 2f
\\1:
\\ sw zero, 0(t0)
\\ addi t0, t0, 4
\\ bltu t0, t1, 1b
\\2:
\\ j zig_main
);
}
|