reorg: split the runtime into library/kernel concern modules (C1)

The one giant `runtime` module (with a `system.zig` that was itself a dumping
ground of unrelated syscalls) is split into directly-importable, flat concern
modules under library/kernel/:

  system-call  ipc  memory  process  thread  time  logging  file-system  service  start
  (+ the device/service clients: device, device-manager, block, display, input)

system.zig is dissolved — its functions moved to their concern home (mmap ->
memory, spawn/kill/exit -> process, sleep/clock -> time, write/klog -> logging,
fs* -> file-system). `memory` merges heap+dma+shared-memory behind one flat API
(memory.allocator/dmaAlloc/sharedCreate/mmap), keeping heap's state and malloc
export single. The memory<->thread dependency cycle (heap needs Thread.Mutex,
thread needs mmap) is broken by having thread allocate its own stack via the raw
mmap syscall, so the module graph is a DAG.

This is the atomic step: all 42 internal cross-imports flip from relative to
module imports at once. `runtime.zig` and `system.zig` become thin re-export
SHIMS so the ~38 consumers keep compiling on `runtime.*` untouched; they migrate
to direct imports in C2, after which the shims are deleted (C5).

zig build + zig build test green; 14 QEMU cases pass (smoke, process,
process-kill, thread-spawn/join, logger, vfs, fat-mount, display-native,
usb-storage, virtio-gpu, device-manager, input, power-button).
This commit is contained in:
Daniel Samson
2026-07-22 22:53:13 +01:00
parent 3e69712b97
commit 60f32ee9ff
21 changed files with 636 additions and 467 deletions
+48
View File
@@ -0,0 +1,48 @@
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
//! physically contiguous, at a physical address the driver knows, uncacheable, and
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
const abi = @import("abi");
const sc = @import("system-call");
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
/// opt-in for specific hardware — see `abi`.
pub const coherent: usize = abi.dma_coherent;
pub const write_combining: usize = abi.dma_write_combining;
pub const below_4g: usize = abi.dma_below_4g;
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
/// to program into the device's descriptor-ring / base registers.
pub const Region = struct {
virtual: usize,
physical: usize,
};
inline fn failed(r: usize) bool {
return r > ~@as(usize, 0) - 4095;
}
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
/// return values — the virtual address in rax, the physical address in rdx — so it
/// needs a hand-written stub.
pub fn alloc(len: usize, flags: usize) ?Region {
var rax: usize = undefined;
var rdx: usize = undefined; // out: physical address
asm volatile ("syscall"
: [rax] "={rax}" (rax),
[rdx] "={rdx}" (rdx),
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
[a0] "{rdi}" (len),
[a1] "{rsi}" (flags),
: .{ .rcx = true, .r11 = true, .memory = true });
if (failed(rax)) return null;
return .{ .virtual = rax, .physical = rdx };
}
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
pub fn free(virtual: usize, len: usize) void {
_ = sc.systemCall2(.dma_free, virtual, len);
}
+215
View File
@@ -0,0 +1,215 @@
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
//! and Zig `std` containers share one heap.
//!
//! The algorithm is a straight port of the kernel's first-fit free list
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
//! split on allocation and coalesced with neighbours on free. The only thing
//! that changes on this side of the system_call boundary is where memory comes from
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
//! itself, and the kernel picks the base address.
//!
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
//! compiles it out and pays nothing, while a threaded one can allocate safely from
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
const std = @import("std");
const builtin = @import("builtin");
const abi = @import("abi");
const sc = @import("system-call");
const Mutex = @import("thread").Thread.Mutex;
const page_size = abi.page_size;
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
var heap_mutex: Mutex = .{};
inline fn lockHeap() void {
if (comptime !builtin.single_threaded) heap_mutex.lock();
}
inline fn unlockHeap() void {
if (comptime !builtin.single_threaded) heap_mutex.unlock();
}
/// A block header, at the start of every block; while free it also links the
/// free list via `next`.
const Block = extern struct {
size: usize, // total block size in bytes, including this header; a multiple of 16
next: ?*Block, // free-list link (only meaningful while free)
};
const header_size = @sizeOf(Block); // 16
const minimum_block = header_size + 16; // smallest block worth splitting off
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
const chunk = 64 * 1024;
var free_list: ?*Block = null;
fn alignUp(value: usize, alignment: usize) usize {
return (value + alignment - 1) & ~(alignment - 1);
}
fn payloadOf(block: *Block) [*]u8 {
return @ptrFromInt(@intFromPtr(block) + header_size);
}
/// Ask the kernel for more pages and add them as a free block. Because each
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
/// grants usually are adjacent). Returns false if the kernel is out of memory.
fn grow(minimum_bytes: usize) bool {
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
const ret = sc.systemCall2(.mmap, bytes, abi.prot_read | abi.prot_write);
if (ret > ~@as(usize, 0) - 4095) return false; // a wrapped -errno lands in the top page
const block: *Block = @ptrFromInt(ret);
block.size = bytes;
insertFree(block); // coalesces if this grant is adjacent to a prior one
return true;
}
/// Insert a block into the address-ordered free list, coalescing with the
/// physically adjacent free blocks on either side.
fn insertFree(block: *Block) void {
var previous: ?*Block = null;
var current = free_list;
while (current) |c| : (current = c.next) {
if (@intFromPtr(c) > @intFromPtr(block)) break;
previous = c;
}
block.next = current;
if (previous) |p| p.next = block else free_list = block;
// Merge forward into `current` if they're contiguous.
if (current) |c| {
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
block.size += c.size;
block.next = c.next;
}
}
// Merge `previous` forward into `block` if they're contiguous.
if (previous) |p| {
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
p.size += block.size;
p.next = block.next;
}
}
}
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
/// across the free-list search and any `grow` (which also touches the free list).
fn rawAlloc(len: usize) ?[*]u8 {
lockHeap();
defer unlockHeap();
const need = alignUp(header_size + len, 16);
var attempts: u32 = 0;
while (attempts < 2) : (attempts += 1) {
var previous: ?*Block = null;
var current = free_list;
while (current) |block| : ({
previous = block;
current = block.next;
}) {
if (block.size < need) continue;
if (block.size >= need + minimum_block) {
// Split: carve `need` off the front, leave the rest free.
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
rest.size = block.size - need;
rest.next = block.next;
if (previous) |p| p.next = rest else free_list = rest;
block.size = need;
} else {
// Take the whole block.
if (previous) |p| p.next = block.next else free_list = block.next;
}
return payloadOf(block);
}
// Nothing fit: grow and try once more.
if (!grow(need)) return null;
}
return null;
}
fn rawFree(ptr: [*]u8) void {
lockHeap();
defer unlockHeap();
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
insertFree(block);
}
// --- C ABI: the global implicit heap ---------------------------------------
// `extern "C"` symbols so future C code links the same malloc/free directly.
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
if (size == 0) return null;
const p = rawAlloc(size) orelse return null;
return @ptrCast(p);
}
export fn free(ptr: ?*anyopaque) callconv(.c) void {
const p = ptr orelse return;
rawFree(@ptrCast(p));
}
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
if (total == 0) return null;
const p = rawAlloc(total) orelse return null;
@memset(p[0..total], 0);
return @ptrCast(p);
}
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
const p = ptr orelse return malloc(size);
if (size == 0) {
rawFree(@ptrCast(p));
return null;
}
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
const old_payload = block.size - header_size;
if (size <= old_payload) return p; // shrink/same: keep the block
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
rawFree(@ptrCast(p));
return @ptrCast(np);
}
// --- std.mem.Allocator interface (same free list) --------------------------
pub fn allocator() std.mem.Allocator {
return .{ .ptr = undefined, .vtable = &vtable };
}
const vtable = std.mem.Allocator.VTable{
.alloc = allocImpl,
.resize = resizeImpl,
.remap = remapImpl,
.free = freeImpl,
};
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
return rawAlloc(len);
}
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
// In-place iff the new payload still fits the current block.
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
return new_len + header_size <= block.size;
}
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
return null;
}
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
rawFree(memory.ptr);
}
+48
View File
@@ -0,0 +1,48 @@
//! library/kernel/memory — the process's memory interface: the heap allocator, DMA-capable
//! buffers, shared-memory regions, and the raw `mmap` grant they all sit on. One flat module
//! (formerly runtime.heap / runtime.dma / runtime.shared_memory, plus the `mmap` wrappers that
//! lived in the system.zig dumping ground). Its private files are heap.zig, dma.zig, and
//! shared-memory.zig — imported only here, so the heap's state and C symbols exist once.
const abi = @import("abi");
const sc = @import("system-call");
const heap = @import("heap.zig");
const dma = @import("dma.zig");
const shared = @import("shared-memory.zig");
// --- the heap: a std.mem.Allocator over a first-fit free list (C malloc/free are also
// exported from heap.zig, compiled once here) ---
pub const allocator = heap.allocator;
// --- the raw grant every allocation sits on ---
pub const PROT_READ: usize = abi.prot_read;
pub const PROT_WRITE: usize = abi.prot_write;
pub const PROT_EXEC: usize = abi.prot_exec;
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable memory and
/// return the base virtual address. On failure returns a value in the top page (`mmapFailed`).
pub fn mmap(len: usize, prot: usize) usize {
return sc.systemCall2(.mmap, len, prot);
}
/// Release a range previously handed out by `mmap`.
pub fn munmap(base: usize, len: usize) usize {
return sc.systemCall2(.munmap, base, len);
}
/// Whether an `mmap` return value is an error (a wrapped -errno lands in the top page).
pub inline fn mmapFailed(ret: usize) bool {
return ret > ~@as(usize, 0) - 4095;
}
// --- DMA-capable buffers: physically contiguous, pinned, uncacheable, physical address known ---
pub const DmaRegion = dma.Region;
pub const dma_coherent = dma.coherent;
pub const dma_write_combining = dma.write_combining;
pub const dma_below_4g = dma.below_4g;
pub const dmaAlloc = dma.alloc;
pub const dmaFree = dma.free;
// --- shared-memory regions: a capability handed to another process over an ipc_call send_cap ---
pub const SharedRegion = shared.Region;
pub const sharedCreate = shared.create;
pub const sharedMap = shared.map;
pub const sharedPhysical = shared.physical;
+57
View File
@@ -0,0 +1,57 @@
//! User-space shared memory: `shared_memory_create` / `shared_memory_map`. A process creates a shareable,
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
//! `shared_memory_map`s it to map the same physical pages. The kernel primitive under the display
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
//! generalization of capability passing from endpoints to memory objects.
const abi = @import("abi");
const sc = @import("system-call");
const ipc = @import("ipc");
inline fn failed(r: usize) bool {
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
}
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
/// another process as an `ipc_call` send_cap.
pub const Region = struct {
ptr: [*]u8,
handle: ipc.Handle,
len: usize,
};
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
/// so this is a hand-written stub like `dma.alloc`.
pub fn create(len: usize) ?Region {
var rax: usize = undefined;
var rdx: usize = undefined; // out: the capability handle
asm volatile ("syscall"
: [rax] "={rax}" (rax),
[rdx] "={rdx}" (rdx),
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shared_memory_create)),
[a0] "{rdi}" (len),
: .{ .rcx = true, .r11 = true, .memory = true });
if (failed(rax)) return null;
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
}
/// Map the shared region named by a capability `handle` this process received (via an
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
/// Returns the pointer, or null on failure.
pub fn map(handle: ipc.Handle) ?[*]u8 {
const r = sc.systemCall1(.shared_memory_map, handle);
if (failed(r)) return null;
return @ptrFromInt(r);
}
/// The guest-physical base of the shared region named by `handle` (which this process must
/// hold a capability for). The region's frames are contiguous, so this single address plus
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
/// `attach_backing`. Returns null on failure.
pub fn physical(handle: ipc.Handle) ?usize {
const r = sc.systemCall1(.shared_memory_physical, handle);
if (failed(r)) return null;
return r;
}