diff options
Diffstat (limited to 'src/io/context.zig')
| -rw-r--r-- | src/io/context.zig | 212 |
1 files changed, 212 insertions, 0 deletions
diff --git a/src/io/context.zig b/src/io/context.zig new file mode 100644 index 0000000..eee8d26 --- /dev/null +++ b/src/io/context.zig @@ -0,0 +1,212 @@ +//! The machine half of the scheduler: what a suspended task is, and how to hand it the CPU. +//! +//! A suspended task is **three words**: `sp`, `fp`, `pc`. Nothing else is stored, and that is not a +//! shortcut - it is the same trick `std.Io.fiber` uses for aarch64, riscv64 and x86_64 +//! (`/usr/lib/zig/std/Io/fiber.zig:7-24`), and it is worth understanding before reading the asm, +//! because "the context switch saves twelve callee-saved registers" is the shape everybody expects +//! and it is not what happens here. +//! +//! The switch declares **every register it does not save as clobbered**. The register allocator +//! then spills whatever is live across the switch onto the switching task's own stack, as part of +//! the calling function's frame, and reloads it when that frame is resumed. So the callee-saved +//! registers *are* saved - by the compiler, into the stack the task already owns, and only the ones +//! that actually hold something. Compiled for this target with `-OReleaseSmall` the spill set +//! around one switch is `ra, s1-s11, fs0-fs11` (verified by reading `-femit-asm` output for +//! riscv32 with `+f`), i.e. 24 words, and it shrinks to nothing in a leaf that holds no live state. +//! A hand-written switch that stores all 24 unconditionally would be both bigger and slower. +//! +//! The exact register set, for the ESP32-P4's rv32imafc: +//! +//! * **saved by this file, in `Context`:** `sp` (x2), `fp` (x8), and the resume address. +//! * **saved by the compiler, because they are clobbered:** `ra` (x1), `t0-t2` (x5-x7), +//! `s1` (x9), `a0` and `a2-a7` (x10, x12-x17), `s2-s11` (x18-x27), `t3-t6` (x28-x31), all 32 +//! `f` registers, and `fflags`/`frm`. `a1` (x11) is the asm's own in/out operand. +//! * **not switched at all, deliberately:** `x0` is hardwired zero. `gp` (x3) is the linker's +//! global pointer - one value for the whole image, never written by a task. `tp` (x4) is the +//! thread pointer; this image has no thread-locals and one hart, so there is nothing per-task +//! to point at. Listing either as a clobber would ask the register allocator to spill a +//! register it can never reload. +//! +//! **First entry differs from a resume in exactly one way: the resume address.** A resume jumps to +//! the label inside the asm block, lands back inside `contextSwitch`, and returns to a frame whose +//! spills are all where the compiler left them. A first entry jumps to a function's first +//! instruction on a stack that contains nothing at all - so on first entry: +//! +//! * `sp` must satisfy the ABI's entry condition (16-byte aligned on RISC-V; 16-byte aligned +//! *minus one word* on x86-64, where a `call` would have pushed a return address); +//! * `fp` is zero, which terminates a frame-pointer walk rather than following garbage; +//! * `ra` is **garbage**, because there is nowhere to return to. The entry function is therefore +//! `noreturn`: it must end in another context switch, and the compiler emits no `ret` for it. +//! On x86-64, where the return address is a stack slot rather than a register, the slot is +//! filled with `returnTrap` so that a mistake is a named panic instead of a wild jump. +//! +//! Nothing is passed to the entry function in a register. There is no way to: on the resumed side +//! the only defined register is the asm's in/out operand, which holds the *switching* task's +//! `Switch` pointer. `p4.zig` reads the current task from the scheduler instead, which it has +//! already set before switching. + +const std = @import("std"); +const builtin = @import("builtin"); + +/// The ESP32-P4 itself. `std.Io.fiber` covers aarch64, riscv64 and x86_64 but not riscv32 +/// (`fiber.zig:1-4`), so the chip's switch is written here and the host's is borrowed from std - +/// which is what lets the scheduler's tests run on the host at all. +pub const rv32 = builtin.cpu.arch == .riscv32; + +pub const supported = rv32 or std.Io.fiber.supported; + +/// A suspended task's whole machine state. 12 bytes on rv32, 24 on the 64-bit hosts. +pub const Context = if (rv32) extern struct { + sp: usize, + fp: usize, + pc: usize, +} else std.Io.fiber.Context; + +/// Layout-identical to `std.Io.fiber.Switch`; a distinct type only so the riscv32 path does not +/// have to name a std type that does not describe it. +pub const Switch = extern struct { old: *Context, new: *Context }; + +/// What the ABI requires of `sp` at a function's first instruction, on every target here. +pub const stack_align = 16; + +/// Save the current machine state into `s.old` and resume the state in `s.new`. +/// +/// Returns when something switches back to `s.old`. Never returns if nothing does - which is the +/// scheduler's problem, not this function's. +pub inline fn contextSwitch(s: *const Switch) void { + if (rv32) { + _ = asm volatile ( + // a1 holds `s` on the way in, and on the way out holds the `Switch` of whoever resumed us. + // The loads must all happen before `sp` moves: after `lw sp, 0(a2)` this frame is gone. + \\ lw a0, 0(a1) + \\ lw a2, 4(a1) + \\ lla a3, 0f + \\ sw sp, 0(a0) + \\ sw fp, 4(a0) + \\ sw a3, 8(a0) + \\ lw sp, 0(a2) + \\ lw fp, 4(a2) + \\ lw a3, 8(a2) + \\ jr a3 + \\0: + : [received] "={a1}" (-> *const Switch), + : [send] "{a1}" (s), + : .{ + .x1 = true, + .x5 = true, + .x6 = true, + .x7 = true, + .x9 = true, + .x10 = true, + .x12 = true, + .x13 = true, + .x14 = true, + .x15 = true, + .x16 = true, + .x17 = true, + .x18 = true, + .x19 = true, + .x20 = true, + .x21 = true, + .x22 = true, + .x23 = true, + .x24 = true, + .x25 = true, + .x26 = true, + .x27 = true, + .x28 = true, + .x29 = true, + .x30 = true, + .x31 = true, + .f0 = true, + .f1 = true, + .f2 = true, + .f3 = true, + .f4 = true, + .f5 = true, + .f6 = true, + .f7 = true, + .f8 = true, + .f9 = true, + .f10 = true, + .f11 = true, + .f12 = true, + .f13 = true, + .f14 = true, + .f15 = true, + .f16 = true, + .f17 = true, + .f18 = true, + .f19 = true, + .f20 = true, + .f21 = true, + .f22 = true, + .f23 = true, + .f24 = true, + .f25 = true, + .f26 = true, + .f27 = true, + .f28 = true, + .f29 = true, + .f30 = true, + .f31 = true, + .fflags = true, + .frm = true, + .memory = true, + }); + } else { + _ = std.Io.fiber.contextSwitch(@ptrCast(s)); + } +} + +/// The `Context` for a task that has never run, which will enter `entry` on the stack ending at +/// `stack_top`. `stack_top` is exclusive, i.e. one past the last writable byte. +/// +/// Writes to the top of the stack on x86-64 (see the file comment); reads nothing. +pub fn initial(stack_top: usize, entry: *const fn () callconv(.c) noreturn) Context { + const top = stack_top & ~@as(usize, stack_align - 1); + return switch (builtin.cpu.arch) { + .x86_64 => sp: { + // System V x86-64 guarantees `rsp % 16 == 0` *before* the `call` that enters a + // function, so a callee's first instruction sees `rsp % 16 == 8` with the return + // address at `[rsp]`. We arrive by `jmp`, so both halves have to be faked, or the + // first 16-byte-aligned spill in the entry function faults. + const sp = top - 8; + @as(*usize, @ptrFromInt(sp)).* = @intFromPtr(&returnTrap); + break :sp .{ .rsp = sp, .rbp = 0, .rip = @intFromPtr(entry) }; + }, + else => .{ .sp = top, .fp = 0, .pc = @intFromPtr(entry) }, + }; +} + +/// How much of the top of a stack `initial` consumes before the entry function's first frame. +pub const initial_stack_overhead: usize = switch (builtin.cpu.arch) { + .x86_64 => 8, + else => 0, +}; + +fn returnTrap() callconv(.c) noreturn { + @panic("io.p4: a task returned from its entry function; the entry function must be noreturn"); +} + +test "initial context points at the entry function on an aligned stack" { + if (!supported) return error.SkipZigTest; + var stack: [256]u8 align(stack_align) = undefined; + const c = initial(@intFromPtr(&stack) + stack.len, returnTrap); + const sp = switch (builtin.cpu.arch) { + .x86_64 => c.rsp, + else => c.sp, + }; + const pc = switch (builtin.cpu.arch) { + .x86_64 => c.rip, + else => c.pc, + }; + try std.testing.expectEqual(@intFromPtr(&returnTrap), pc); + try std.testing.expect(sp <= @intFromPtr(&stack) + stack.len); + try std.testing.expect(sp > @intFromPtr(&stack)); + // The entry condition the ABI states, per target. + switch (builtin.cpu.arch) { + .x86_64 => try std.testing.expectEqual(@as(usize, 8), sp % stack_align), + else => try std.testing.expectEqual(@as(usize, 0), sp % stack_align), + } +} |
