summaryrefslogtreecommitdiff
path: root/src/io/context.zig
diff options
context:
space:
mode:
Diffstat (limited to 'src/io/context.zig')
-rw-r--r--src/io/context.zig212
1 files changed, 212 insertions, 0 deletions
diff --git a/src/io/context.zig b/src/io/context.zig
new file mode 100644
index 0000000..eee8d26
--- /dev/null
+++ b/src/io/context.zig
@@ -0,0 +1,212 @@
+//! The machine half of the scheduler: what a suspended task is, and how to hand it the CPU.
+//!
+//! A suspended task is **three words**: `sp`, `fp`, `pc`. Nothing else is stored, and that is not a
+//! shortcut - it is the same trick `std.Io.fiber` uses for aarch64, riscv64 and x86_64
+//! (`/usr/lib/zig/std/Io/fiber.zig:7-24`), and it is worth understanding before reading the asm,
+//! because "the context switch saves twelve callee-saved registers" is the shape everybody expects
+//! and it is not what happens here.
+//!
+//! The switch declares **every register it does not save as clobbered**. The register allocator
+//! then spills whatever is live across the switch onto the switching task's own stack, as part of
+//! the calling function's frame, and reloads it when that frame is resumed. So the callee-saved
+//! registers *are* saved - by the compiler, into the stack the task already owns, and only the ones
+//! that actually hold something. Compiled for this target with `-OReleaseSmall` the spill set
+//! around one switch is `ra, s1-s11, fs0-fs11` (verified by reading `-femit-asm` output for
+//! riscv32 with `+f`), i.e. 24 words, and it shrinks to nothing in a leaf that holds no live state.
+//! A hand-written switch that stores all 24 unconditionally would be both bigger and slower.
+//!
+//! The exact register set, for the ESP32-P4's rv32imafc:
+//!
+//! * **saved by this file, in `Context`:** `sp` (x2), `fp` (x8), and the resume address.
+//! * **saved by the compiler, because they are clobbered:** `ra` (x1), `t0-t2` (x5-x7),
+//! `s1` (x9), `a0` and `a2-a7` (x10, x12-x17), `s2-s11` (x18-x27), `t3-t6` (x28-x31), all 32
+//! `f` registers, and `fflags`/`frm`. `a1` (x11) is the asm's own in/out operand.
+//! * **not switched at all, deliberately:** `x0` is hardwired zero. `gp` (x3) is the linker's
+//! global pointer - one value for the whole image, never written by a task. `tp` (x4) is the
+//! thread pointer; this image has no thread-locals and one hart, so there is nothing per-task
+//! to point at. Listing either as a clobber would ask the register allocator to spill a
+//! register it can never reload.
+//!
+//! **First entry differs from a resume in exactly one way: the resume address.** A resume jumps to
+//! the label inside the asm block, lands back inside `contextSwitch`, and returns to a frame whose
+//! spills are all where the compiler left them. A first entry jumps to a function's first
+//! instruction on a stack that contains nothing at all - so on first entry:
+//!
+//! * `sp` must satisfy the ABI's entry condition (16-byte aligned on RISC-V; 16-byte aligned
+//! *minus one word* on x86-64, where a `call` would have pushed a return address);
+//! * `fp` is zero, which terminates a frame-pointer walk rather than following garbage;
+//! * `ra` is **garbage**, because there is nowhere to return to. The entry function is therefore
+//! `noreturn`: it must end in another context switch, and the compiler emits no `ret` for it.
+//! On x86-64, where the return address is a stack slot rather than a register, the slot is
+//! filled with `returnTrap` so that a mistake is a named panic instead of a wild jump.
+//!
+//! Nothing is passed to the entry function in a register. There is no way to: on the resumed side
+//! the only defined register is the asm's in/out operand, which holds the *switching* task's
+//! `Switch` pointer. `p4.zig` reads the current task from the scheduler instead, which it has
+//! already set before switching.
+
+const std = @import("std");
+const builtin = @import("builtin");
+
+/// The ESP32-P4 itself. `std.Io.fiber` covers aarch64, riscv64 and x86_64 but not riscv32
+/// (`fiber.zig:1-4`), so the chip's switch is written here and the host's is borrowed from std -
+/// which is what lets the scheduler's tests run on the host at all.
+pub const rv32 = builtin.cpu.arch == .riscv32;
+
+pub const supported = rv32 or std.Io.fiber.supported;
+
+/// A suspended task's whole machine state. 12 bytes on rv32, 24 on the 64-bit hosts.
+pub const Context = if (rv32) extern struct {
+ sp: usize,
+ fp: usize,
+ pc: usize,
+} else std.Io.fiber.Context;
+
+/// Layout-identical to `std.Io.fiber.Switch`; a distinct type only so the riscv32 path does not
+/// have to name a std type that does not describe it.
+pub const Switch = extern struct { old: *Context, new: *Context };
+
+/// What the ABI requires of `sp` at a function's first instruction, on every target here.
+pub const stack_align = 16;
+
+/// Save the current machine state into `s.old` and resume the state in `s.new`.
+///
+/// Returns when something switches back to `s.old`. Never returns if nothing does - which is the
+/// scheduler's problem, not this function's.
+pub inline fn contextSwitch(s: *const Switch) void {
+ if (rv32) {
+ _ = asm volatile (
+ // a1 holds `s` on the way in, and on the way out holds the `Switch` of whoever resumed us.
+ // The loads must all happen before `sp` moves: after `lw sp, 0(a2)` this frame is gone.
+ \\ lw a0, 0(a1)
+ \\ lw a2, 4(a1)
+ \\ lla a3, 0f
+ \\ sw sp, 0(a0)
+ \\ sw fp, 4(a0)
+ \\ sw a3, 8(a0)
+ \\ lw sp, 0(a2)
+ \\ lw fp, 4(a2)
+ \\ lw a3, 8(a2)
+ \\ jr a3
+ \\0:
+ : [received] "={a1}" (-> *const Switch),
+ : [send] "{a1}" (s),
+ : .{
+ .x1 = true,
+ .x5 = true,
+ .x6 = true,
+ .x7 = true,
+ .x9 = true,
+ .x10 = true,
+ .x12 = true,
+ .x13 = true,
+ .x14 = true,
+ .x15 = true,
+ .x16 = true,
+ .x17 = true,
+ .x18 = true,
+ .x19 = true,
+ .x20 = true,
+ .x21 = true,
+ .x22 = true,
+ .x23 = true,
+ .x24 = true,
+ .x25 = true,
+ .x26 = true,
+ .x27 = true,
+ .x28 = true,
+ .x29 = true,
+ .x30 = true,
+ .x31 = true,
+ .f0 = true,
+ .f1 = true,
+ .f2 = true,
+ .f3 = true,
+ .f4 = true,
+ .f5 = true,
+ .f6 = true,
+ .f7 = true,
+ .f8 = true,
+ .f9 = true,
+ .f10 = true,
+ .f11 = true,
+ .f12 = true,
+ .f13 = true,
+ .f14 = true,
+ .f15 = true,
+ .f16 = true,
+ .f17 = true,
+ .f18 = true,
+ .f19 = true,
+ .f20 = true,
+ .f21 = true,
+ .f22 = true,
+ .f23 = true,
+ .f24 = true,
+ .f25 = true,
+ .f26 = true,
+ .f27 = true,
+ .f28 = true,
+ .f29 = true,
+ .f30 = true,
+ .f31 = true,
+ .fflags = true,
+ .frm = true,
+ .memory = true,
+ });
+ } else {
+ _ = std.Io.fiber.contextSwitch(@ptrCast(s));
+ }
+}
+
+/// The `Context` for a task that has never run, which will enter `entry` on the stack ending at
+/// `stack_top`. `stack_top` is exclusive, i.e. one past the last writable byte.
+///
+/// Writes to the top of the stack on x86-64 (see the file comment); reads nothing.
+pub fn initial(stack_top: usize, entry: *const fn () callconv(.c) noreturn) Context {
+ const top = stack_top & ~@as(usize, stack_align - 1);
+ return switch (builtin.cpu.arch) {
+ .x86_64 => sp: {
+ // System V x86-64 guarantees `rsp % 16 == 0` *before* the `call` that enters a
+ // function, so a callee's first instruction sees `rsp % 16 == 8` with the return
+ // address at `[rsp]`. We arrive by `jmp`, so both halves have to be faked, or the
+ // first 16-byte-aligned spill in the entry function faults.
+ const sp = top - 8;
+ @as(*usize, @ptrFromInt(sp)).* = @intFromPtr(&returnTrap);
+ break :sp .{ .rsp = sp, .rbp = 0, .rip = @intFromPtr(entry) };
+ },
+ else => .{ .sp = top, .fp = 0, .pc = @intFromPtr(entry) },
+ };
+}
+
+/// How much of the top of a stack `initial` consumes before the entry function's first frame.
+pub const initial_stack_overhead: usize = switch (builtin.cpu.arch) {
+ .x86_64 => 8,
+ else => 0,
+};
+
+fn returnTrap() callconv(.c) noreturn {
+ @panic("io.p4: a task returned from its entry function; the entry function must be noreturn");
+}
+
+test "initial context points at the entry function on an aligned stack" {
+ if (!supported) return error.SkipZigTest;
+ var stack: [256]u8 align(stack_align) = undefined;
+ const c = initial(@intFromPtr(&stack) + stack.len, returnTrap);
+ const sp = switch (builtin.cpu.arch) {
+ .x86_64 => c.rsp,
+ else => c.sp,
+ };
+ const pc = switch (builtin.cpu.arch) {
+ .x86_64 => c.rip,
+ else => c.pc,
+ };
+ try std.testing.expectEqual(@intFromPtr(&returnTrap), pc);
+ try std.testing.expect(sp <= @intFromPtr(&stack) + stack.len);
+ try std.testing.expect(sp > @intFromPtr(&stack));
+ // The entry condition the ABI states, per target.
+ switch (builtin.cpu.arch) {
+ .x86_64 => try std.testing.expectEqual(@as(usize, 8), sp % stack_align),
+ else => try std.testing.expectEqual(@as(usize, 0), sp % stack_align),
+ }
+}