1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
|
//! The machine half of the scheduler: what a suspended task is, and how to hand it the CPU.
//!
//! A suspended task is **three words**: `sp`, `fp`, `pc`. Nothing else is stored, and that is not a
//! shortcut - it is the same trick `std.Io.fiber` uses for aarch64, riscv64 and x86_64
//! (`/usr/lib/zig/std/Io/fiber.zig:7-24`), and it is worth understanding before reading the asm,
//! because "the context switch saves twelve callee-saved registers" is the shape everybody expects
//! and it is not what happens here.
//!
//! The switch declares **every register it does not save as clobbered**. The register allocator
//! then spills whatever is live across the switch onto the switching task's own stack, as part of
//! the calling function's frame, and reloads it when that frame is resumed. So the callee-saved
//! registers *are* saved - by the compiler, into the stack the task already owns, and only the ones
//! that actually hold something. Compiled for this target with `-OReleaseSmall` the spill set
//! around one switch is `ra, s1-s11, fs0-fs11` (verified by reading `-femit-asm` output for
//! riscv32 with `+f`), i.e. 24 words, and it shrinks to nothing in a leaf that holds no live state.
//! A hand-written switch that stores all 24 unconditionally would be both bigger and slower.
//!
//! The exact register set, for the ESP32-P4's rv32imafc:
//!
//! * **saved by this file, in `Context`:** `sp` (x2), `fp` (x8), and the resume address.
//! * **saved by the compiler, because they are clobbered:** `ra` (x1), `t0-t2` (x5-x7),
//! `s1` (x9), `a0` and `a2-a7` (x10, x12-x17), `s2-s11` (x18-x27), `t3-t6` (x28-x31), all 32
//! `f` registers, and `fflags`/`frm`. `a1` (x11) is the asm's own in/out operand.
//! * **not switched at all, deliberately:** `x0` is hardwired zero. `gp` (x3) is the linker's
//! global pointer - one value for the whole image, never written by a task. `tp` (x4) is the
//! thread pointer; this image has no thread-locals and one hart, so there is nothing per-task
//! to point at. Listing either as a clobber would ask the register allocator to spill a
//! register it can never reload.
//!
//! **First entry differs from a resume in exactly one way: the resume address.** A resume jumps to
//! the label inside the asm block, lands back inside `contextSwitch`, and returns to a frame whose
//! spills are all where the compiler left them. A first entry jumps to a function's first
//! instruction on a stack that contains nothing at all - so on first entry:
//!
//! * `sp` must satisfy the ABI's entry condition (16-byte aligned on RISC-V; 16-byte aligned
//! *minus one word* on x86-64, where a `call` would have pushed a return address);
//! * `fp` is zero, which terminates a frame-pointer walk rather than following garbage;
//! * `ra` is **garbage**, because there is nowhere to return to. The entry function is therefore
//! `noreturn`: it must end in another context switch, and the compiler emits no `ret` for it.
//! On x86-64, where the return address is a stack slot rather than a register, the slot is
//! filled with `returnTrap` so that a mistake is a named panic instead of a wild jump.
//!
//! Nothing is passed to the entry function in a register. There is no way to: on the resumed side
//! the only defined register is the asm's in/out operand, which holds the *switching* task's
//! `Switch` pointer. `p4.zig` reads the current task from the scheduler instead, which it has
//! already set before switching.
const std = @import("std");
const builtin = @import("builtin");
/// The ESP32-P4 itself. `std.Io.fiber` covers aarch64, riscv64 and x86_64 but not riscv32
/// (`fiber.zig:1-4`), so the chip's switch is written here and the host's is borrowed from std -
/// which is what lets the scheduler's tests run on the host at all.
pub const rv32 = builtin.cpu.arch == .riscv32;
pub const supported = rv32 or std.Io.fiber.supported;
/// A suspended task's whole machine state. 12 bytes on rv32, 24 on the 64-bit hosts.
pub const Context = if (rv32) extern struct {
sp: usize,
fp: usize,
pc: usize,
} else std.Io.fiber.Context;
/// Layout-identical to `std.Io.fiber.Switch`; a distinct type only so the riscv32 path does not
/// have to name a std type that does not describe it.
pub const Switch = extern struct { old: *Context, new: *Context };
/// What the ABI requires of `sp` at a function's first instruction, on every target here.
pub const stack_align = 16;
/// Save the current machine state into `s.old` and resume the state in `s.new`.
///
/// Returns when something switches back to `s.old`. Never returns if nothing does - which is the
/// scheduler's problem, not this function's.
pub inline fn contextSwitch(s: *const Switch) void {
if (rv32) {
_ = asm volatile (
// a1 holds `s` on the way in, and on the way out holds the `Switch` of whoever resumed us.
// The loads must all happen before `sp` moves: after `lw sp, 0(a2)` this frame is gone.
\\ lw a0, 0(a1)
\\ lw a2, 4(a1)
\\ lla a3, 0f
\\ sw sp, 0(a0)
\\ sw fp, 4(a0)
\\ sw a3, 8(a0)
\\ lw sp, 0(a2)
\\ lw fp, 4(a2)
\\ lw a3, 8(a2)
\\ jr a3
\\0:
: [received] "={a1}" (-> *const Switch),
: [send] "{a1}" (s),
: .{
.x1 = true,
.x5 = true,
.x6 = true,
.x7 = true,
.x9 = true,
.x10 = true,
.x12 = true,
.x13 = true,
.x14 = true,
.x15 = true,
.x16 = true,
.x17 = true,
.x18 = true,
.x19 = true,
.x20 = true,
.x21 = true,
.x22 = true,
.x23 = true,
.x24 = true,
.x25 = true,
.x26 = true,
.x27 = true,
.x28 = true,
.x29 = true,
.x30 = true,
.x31 = true,
.f0 = true,
.f1 = true,
.f2 = true,
.f3 = true,
.f4 = true,
.f5 = true,
.f6 = true,
.f7 = true,
.f8 = true,
.f9 = true,
.f10 = true,
.f11 = true,
.f12 = true,
.f13 = true,
.f14 = true,
.f15 = true,
.f16 = true,
.f17 = true,
.f18 = true,
.f19 = true,
.f20 = true,
.f21 = true,
.f22 = true,
.f23 = true,
.f24 = true,
.f25 = true,
.f26 = true,
.f27 = true,
.f28 = true,
.f29 = true,
.f30 = true,
.f31 = true,
.fflags = true,
.frm = true,
.memory = true,
});
} else {
_ = std.Io.fiber.contextSwitch(@ptrCast(s));
}
}
/// The `Context` for a task that has never run, which will enter `entry` on the stack ending at
/// `stack_top`. `stack_top` is exclusive, i.e. one past the last writable byte.
///
/// Writes to the top of the stack on x86-64 (see the file comment); reads nothing.
pub fn initial(stack_top: usize, entry: *const fn () callconv(.c) noreturn) Context {
const top = stack_top & ~@as(usize, stack_align - 1);
return switch (builtin.cpu.arch) {
.x86_64 => sp: {
// System V x86-64 guarantees `rsp % 16 == 0` *before* the `call` that enters a
// function, so a callee's first instruction sees `rsp % 16 == 8` with the return
// address at `[rsp]`. We arrive by `jmp`, so both halves have to be faked, or the
// first 16-byte-aligned spill in the entry function faults.
const sp = top - 8;
@as(*usize, @ptrFromInt(sp)).* = @intFromPtr(&returnTrap);
break :sp .{ .rsp = sp, .rbp = 0, .rip = @intFromPtr(entry) };
},
else => .{ .sp = top, .fp = 0, .pc = @intFromPtr(entry) },
};
}
/// How much of the top of a stack `initial` consumes before the entry function's first frame.
pub const initial_stack_overhead: usize = switch (builtin.cpu.arch) {
.x86_64 => 8,
else => 0,
};
fn returnTrap() callconv(.c) noreturn {
@panic("io.p4: a task returned from its entry function; the entry function must be noreturn");
}
test "initial context points at the entry function on an aligned stack" {
if (!supported) return error.SkipZigTest;
var stack: [256]u8 align(stack_align) = undefined;
const c = initial(@intFromPtr(&stack) + stack.len, returnTrap);
const sp = switch (builtin.cpu.arch) {
.x86_64 => c.rsp,
else => c.sp,
};
const pc = switch (builtin.cpu.arch) {
.x86_64 => c.rip,
else => c.pc,
};
try std.testing.expectEqual(@intFromPtr(&returnTrap), pc);
try std.testing.expect(sp <= @intFromPtr(&stack) + stack.len);
try std.testing.expect(sp > @intFromPtr(&stack));
// The entry condition the ABI states, per target.
switch (builtin.cpu.arch) {
.x86_64 => try std.testing.expectEqual(@as(usize, 8), sp % stack_align),
else => try std.testing.expectEqual(@as(usize, 0), sp % stack_align),
}
}
|