1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
|
//! The test suite that runs ON THE DIE.
//!
//! WHY THIS EXISTS. Every serious bug this port has produced was invisible to a host test, and two
//! of them were invisible for months. `std.mem.eql` compares a byte at a time on this target and
//! several times faster on the host, so the firmware's largest read was three times slower than it
//! needed to be and nothing on a laptop could tell. A lone ESC resolves to the Escape key, which is
//! right when a kernel hands over a whole escape sequence and wrong when a 115200 line hands over
//! one byte every 87 microseconds. A full transmit FIFO stopped anything draining the receiver, and
//! the FIFO depth is a hardware number. None of those is a logic error you can reason your way to
//! from a host: they are properties of THIS chip, THIS clock and THIS wire.
//!
//! So the checks below are chosen by one rule: a check belongs here only if the die can answer it
//! and a host cannot. Anything that is pure logic - the ring's wrap-around, the mouse coalescer's
//! ordering - already has a deterministic host test in `zig build test`, which is faster, needs no
//! hardware, and is where such a thing belongs. Duplicating those here would make this suite longer
//! and no stronger.
//!
//! Not a `zig test` binary, deliberately. Zig's test runner wants an OS, and `std.testing.allocator`
//! is a debug allocator over the page allocator, which on freestanding is either a compile error or
//! a lie. A hand-rolled harness is thirty lines and answers to nobody.
//!
//! Run with: zig build selftest
//! or: zig build -Dapp=examples/selftest.zig run
const std = @import("std");
const soc = @import("soc");
const hal = @import("hal");
const config = @import("config");
const heapmod = @import("heap");
const input_rescue = @import("input_rescue");
/// Reset entry. Identical in shape to `src/main.zig`'s and for the same reasons: the bootloader hands
/// over with an unspecified stack pointer and the FPU off, so set `mstatus.FS`, establish a stack,
/// clear `.bss`, and jump.
///
/// Leaving this out is what made the first draft of this file unbuildable, and the failure said
/// nothing useful: `-fentry=_start` found no such symbol, `--gc-sections` then discarded every
/// function as unreachable, and the image builder reported `NotTwoMappedSegments` because
/// `.flash.text` had nothing left in it. `zig build layout` now prints `entry 0x0` for exactly that,
/// which is the same diagnosis in one line.
export fn _start() linksection(".text.entry") callconv(.naked) noreturn {
asm volatile (
\\ li t0, 1 << 13
\\ csrs mstatus, t0
\\ la sp, __stack_top
\\ mv fp, sp
\\ la t0, __bss_start
\\ la t1, __bss_end
\\ bgeu t0, t1, 2f
\\1:
\\ sw zero, 0(t0)
\\ addi t0, t0, 4
\\ bltu t0, t1, 1b
\\2:
\\ j zig_main
);
}
pub const panic = std.debug.FullPanic(struct {
fn call(msg: []const u8, first_trace_addr: ?usize) noreturn {
// `msg` is a slice and carries no terminator - std's own panics are formatted into a buffer -
// so it goes out with a length rather than through a `%s` that would read past the end of it.
soc.rom.print("\r\nMARK SELFTEST_PANIC at 0x%08x: ", .{@as(u32, @truncate(first_trace_addr orelse 0))});
hal.uart.Uart.init(0).write(msg);
// A panic is a FAILED RUN, and the host is watching for the summary line. Without this the
// run looks like a board that never answered, which is a different diagnosis entirely.
soc.rom.print("\r\nMARK SELFTEST DONE pass=%u fail=%u\r\n", .{ passed, failed + 1 });
while (true) {}
}
}.call);
var passed: u32 = 0;
var failed: u32 = 0;
/// One claim about the silicon. Printed either way: a suite that only speaks up when it fails gives
/// no way to tell "all good" from "never ran", and on a board that difference matters.
fn check(name: [*:0]const u8, ok: bool) void {
if (ok) {
passed += 1;
soc.rom.print("MARK SELFTEST ok %s\r\n", .{name});
} else {
failed += 1;
soc.rom.print("MARK SELFTEST FAIL %s\r\n", .{name});
}
}
/// Word-at-a-time equality, the same shape the editor's frame diff uses.
fn sameBytesWordwise(a: []const u8, b: []const u8) bool {
if (a.len != b.len) return false;
if ((@intFromPtr(a.ptr) | @intFromPtr(b.ptr)) & 3 == 0) {
const n = a.len / 4;
const wa: [*]align(4) const u32 = @ptrCast(@alignCast(a.ptr));
const wb: [*]align(4) const u32 = @ptrCast(@alignCast(b.ptr));
for (wa[0..n], wb[0..n]) |x, y| {
if (x != y) return false;
}
return std.mem.eql(u8, a[n * 4 ..], b[n * 4 ..]);
}
return std.mem.eql(u8, a, b);
}
/// A UART with a receive FIFO that loses whatever arrives into a full one, as the hardware does.
/// `pub` on the methods because `input_rescue` is a separate module here and reaches them by duck
/// typing across it.
const FakePort = struct {
tx_cap: u32,
tx_used: u32 = 0,
ticks: u32 = 0,
sent: u32 = 0,
incoming: []const u8,
delivered: usize = 0,
rx: [8]u8 = undefined,
rx_head: usize = 0,
rx_len: usize = 0,
lost: u32 = 0,
fn tick(p: *FakePort) void {
p.ticks += 1;
if (p.ticks % 4 == 0 and p.tx_used > 0) p.tx_used -= 1;
if (p.delivered < p.incoming.len) {
const byte = p.incoming[p.delivered];
p.delivered += 1;
if (p.rx_len == p.rx.len) {
p.lost += 1;
} else {
p.rx[(p.rx_head + p.rx_len) % p.rx.len] = byte;
p.rx_len += 1;
}
}
}
pub fn txFree(p: *FakePort) u32 {
p.tick();
return p.tx_cap - p.tx_used;
}
pub fn pushByte(p: *FakePort, _: u8) void {
p.sent += 1;
p.tx_used += 1;
}
pub fn rxCount(p: *FakePort) u32 {
return @intCast(p.rx_len);
}
pub fn popByte(p: *FakePort) u8 {
const byte = p.rx[p.rx_head];
p.rx_head = (p.rx_head + 1) % p.rx.len;
p.rx_len -= 1;
return byte;
}
};
/// Backing store for the heap checks. Static, because the point is to exercise the allocator on real
/// L2MEM rather than to find out where a stack array happens to land.
var heap_area: [64 * 1024]u8 align(16) = undefined;
export fn zig_main() noreturn {
// THE CLOCK RAISE, performed here rather than assumed, which is what turns the frequency check
// at the end into a test of `setCpuFreq` instead of a tautology. The first version of this file
// read `config.cpu_mhz` and compared the die against it without ever setting it, so
// `-Dcpu-mhz=360` failed with `khz=90001 want=360000` - the check was right and the expectation
// was wrong. Same call, and the same order, as `src/pardes/app.zig`.
if (config.cpu_mhz != 90) hal.clkrst.setCpuFreq(switch (config.cpu_mhz) {
360 => .mhz360,
else => .mhz90,
});
hal.systimer.init();
soc.rom.print("\r\nMARK SELFTEST_START cpu_mhz=%u\r\n", .{@as(u32, config.cpu_mhz)});
// ------------------------------------------------ 1. the memory model the memory words assume
//
// `Peek`, `Poke` and `Hexdump` reach the bus through `*allowzero volatile` pointers and refuse an
// unaligned word. Both halves are claims about this core, and neither is checkable on a host.
{
const cell: *volatile u32 = @ptrCast(@alignCast(&heap_area[0]));
cell.* = 0xdeadbeef;
check("an aligned word round-trips through a volatile pointer", cell.* == 0xdeadbeef);
// Every byte offset in a word, readable and writable: this is what `Hexdump` does, and it is
// why `Hexdump` needs no alignment while `Peek` does.
var all_offsets_ok = true;
for (0..4) |i| {
const at: *volatile u8 = @ptrCast(&heap_area[16 + i]);
at.* = @intCast(0xa0 + i);
if (at.* != 0xa0 + @as(u8, @intCast(i))) all_offsets_ok = false;
}
check("a byte at every offset in a word round-trips", all_offsets_ok);
// THE VOLATILE PROMISE. Two reads of a running counter must be two reads. Were the optimiser
// allowed to fold them, `Peek` would print one value twice for a register that had changed,
// which is the one thing a memory word must never do.
const first = hal.systimer.micros(.unit0) orelse 0;
var spin: u32 = 0;
while (spin < 4000) : (spin += 1) asm volatile ("" ::: .{ .memory = true });
const second = hal.systimer.micros(.unit0) orelse 0;
check("two reads of a live counter are two reads", second != first);
check("and that counter runs forwards", second > first);
}
// --------------------------------------- 2. word-wise equality, on THIS instruction set
//
// The editor's frame diff compares rows a `u32` at a time because `std.mem.eql` compares a byte
// at a time here: 223 us against 66 for the same answer. "The same answer" is the part that has
// to hold on the target rather than on the host, so it is checked against `std.mem.eql` itself,
// at every difference position, at both alignments, and at lengths that are not multiples of 4.
{
var a: [64]u8 align(4) = undefined;
var b: [64]u8 align(4) = undefined;
for (&a, 0..) |*slot, i| slot.* = @intCast(i);
@memcpy(&b, &a);
var agree = true;
for (0..a.len) |len| {
if (sameBytesWordwise(a[0..len], b[0..len]) != std.mem.eql(u8, a[0..len], b[0..len])) agree = false;
}
check("aligned equality agrees with std.mem.eql at every length", agree);
agree = true;
for (0..a.len) |i| {
b[i] ^= 0xff;
if (sameBytesWordwise(&a, &b) != std.mem.eql(u8, &a, &b)) agree = false;
if (sameBytesWordwise(a[0..33], b[0..33]) != std.mem.eql(u8, a[0..33], b[0..33])) agree = false;
b[i] ^= 0xff;
}
check("a difference at any position is found, as std.mem.eql finds it", agree);
agree = true;
// UNALIGNED, which is why `sameBytes` tests alignment at RUNTIME: `Cell` is all u8 fields, so
// whether a row starts on a word boundary belongs to the allocator and not to the type.
for (1..4) |off| {
const ua = a[off..];
const ub = b[off..];
if (sameBytesWordwise(ua, ub) != std.mem.eql(u8, ua, ub)) agree = false;
b[off + 5] ^= 0xff;
if (sameBytesWordwise(ua, ub) != std.mem.eql(u8, ua, ub)) agree = false;
b[off + 5] ^= 0xff;
}
check("unaligned spans fall back and still agree", agree);
}
// ------------------------------------------------- 3. the rescue, on the real codegen
//
// The logic has a host test. What that cannot say is whether it behaves the same compiled for
// this core at this optimisation level, which is a question only a board answers.
{
const typed = "the quick brown fox jumps over the lazy dog, and then some more besides";
var port: FakePort = .{ .tx_cap = 2, .incoming = typed };
var ring: input_rescue.Ring = .{};
const frame = "\x1b[1;1H" ++ "x" ** 300;
const abandoned = input_rescue.pump(&port, &ring, frame, 1_000_000);
check("the frame went out whole", abandoned == 0 and port.sent == frame.len);
check("the receive FIFO never overflowed", port.lost == 0);
check("the ring dropped nothing", ring.dropped == 0);
var got: [128]u8 = undefined;
var n = ring.pop(&got);
while (port.rxCount() > 0 and n < got.len) : (n += 1) got[n] = port.popByte();
check("every rescued byte, in order", n == typed.len and std.mem.eql(u8, got[0..n], typed));
}
// --------------------------------------------------- 4. the allocator on real L2MEM
//
// The editor's whole geometry ceiling is an allocator question, and this is the allocator, on the
// memory it actually runs in rather than on a host's malloc.
{
var h = heapmod.Heap.init(heap_area[0..]);
const gpa = h.allocator();
const one = gpa.alloc(u32, 256) catch null;
check("a modest allocation succeeds", one != null);
if (one) |slice| {
check("and is aligned for its element", @intFromPtr(slice.ptr) % @alignOf(u32) == 0);
for (slice, 0..) |*slot, i| slot.* = @intCast(i * 7);
var intact = true;
for (slice, 0..) |slot, i| {
if (slot != i * 7) intact = false;
}
check("and holds what was written to it", intact);
gpa.free(slice);
}
// FREE THEN REUSE. A heap that cannot hand the same bytes back is a heap that runs out, which
// on this board is the difference between a 40x12 grid and an 80x24 one.
var churn_ok = true;
for (0..64) |_| {
const block = gpa.alloc(u8, 1024) catch {
churn_ok = false;
break;
};
gpa.free(block);
}
check("a kilobyte can be taken and returned repeatedly", churn_ok);
// AND IT REFUSES CLEANLY. An allocator that returns garbage instead of an error when it is
// out is the failure mode that cost an afternoon during bring-up.
const absurd = gpa.alloc(u8, heap_area.len * 4);
check("an impossible allocation returns an error", absurd == error.OutOfMemory);
}
// ------------------------------------------ 5. the clock, which everything divides by
//
// Every cycle count this firmware reports is divided by the configured frequency somewhere, and
// the systimer is clocked from the crystal rather than from the CPU - which is exactly what makes
// it a reference the CPU cannot flatter. The 90-to-360 MHz raise rested on this comparison.
{
const t0 = hal.systimer.micros(.unit0) orelse 0;
const c0 = soc.cycles();
while ((hal.systimer.micros(.unit0) orelse 0) -% t0 < 20_000) {}
const us = (hal.systimer.micros(.unit0) orelse 0) -% t0;
const cy = soc.cycles() - c0;
const khz: u32 = if (us > 0) @intCast(cy * 1000 / us) else 0;
const want: u32 = @as(u32, config.cpu_mhz) * 1000;
// Two percent: far wider than either clock's error, far narrower than the 4x a wrong divider
// would produce.
const slack = want / 50;
soc.rom.print("MARK SELFTEST_CLOCK khz=%u want=%u\r\n", .{ khz, want });
check("the cycle counter and the systimer agree on the CPU frequency", khz > want - slack and khz < want + slack);
}
soc.rom.print("MARK SELFTEST DONE pass=%u fail=%u\r\n", .{ passed, failed });
while (true) {}
}
|