reorg: split the runtime into library/kernel concern modules (C1)
The one giant `runtime` module (with a `system.zig` that was itself a dumping ground of unrelated syscalls) is split into directly-importable, flat concern modules under library/kernel/: system-call ipc memory process thread time logging file-system service start (+ the device/service clients: device, device-manager, block, display, input) system.zig is dissolved — its functions moved to their concern home (mmap -> memory, spawn/kill/exit -> process, sleep/clock -> time, write/klog -> logging, fs* -> file-system). `memory` merges heap+dma+shared-memory behind one flat API (memory.allocator/dmaAlloc/sharedCreate/mmap), keeping heap's state and malloc export single. The memory<->thread dependency cycle (heap needs Thread.Mutex, thread needs mmap) is broken by having thread allocate its own stack via the raw mmap syscall, so the module graph is a DAG. This is the atomic step: all 42 internal cross-imports flip from relative to module imports at once. `runtime.zig` and `system.zig` become thin re-export SHIMS so the ~38 consumers keep compiling on `runtime.*` untouched; they migrate to direct imports in C2, after which the shims are deleted (C5). zig build + zig build test green; 14 QEMU cases pass (smoke, process, process-kill, thread-spawn/join, logger, vfs, fat-mount, display-native, usb-storage, virtio-gpu, device-manager, input, power-button).
This commit is contained in:
@@ -0,0 +1,48 @@
|
||||
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
|
||||
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||
/// opt-in for specific hardware — see `abi`.
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
pub fn free(virtual: usize, len: usize) void {
|
||||
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||
}
|
||||
@@ -0,0 +1,215 @@
|
||||
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||
//! and Zig `std` containers share one heap.
|
||||
//!
|
||||
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||
//! that changes on this side of the system_call boundary is where memory comes from
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
|
||||
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
|
||||
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
|
||||
//! compiles it out and pays nothing, while a threaded one can allocate safely from
|
||||
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
|
||||
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
|
||||
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const Mutex = @import("thread").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
|
||||
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
|
||||
var heap_mutex: Mutex = .{};
|
||||
|
||||
inline fn lockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.lock();
|
||||
}
|
||||
inline fn unlockHeap() void {
|
||||
if (comptime !builtin.single_threaded) heap_mutex.unlock();
|
||||
}
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||
const chunk = 64 * 1024;
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = sc.systemCall2(.mmap, bytes, abi.prot_read | abi.prot_write);
|
||||
if (ret > ~@as(usize, 0) - 4095) return false; // a wrapped -errno lands in the top page
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
|
||||
/// across the free-list search and any `grow` (which also touches the free list).
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
lockHeap();
|
||||
defer unlockHeap();
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- C ABI: the global implicit heap ---------------------------------------
|
||||
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||
|
||||
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||
if (size == 0) return null;
|
||||
const p = rawAlloc(size) orelse return null;
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||
const p = ptr orelse return;
|
||||
rawFree(@ptrCast(p));
|
||||
}
|
||||
|
||||
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||
if (total == 0) return null;
|
||||
const p = rawAlloc(total) orelse return null;
|
||||
@memset(p[0..total], 0);
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||
const p = ptr orelse return malloc(size);
|
||||
if (size == 0) {
|
||||
rawFree(@ptrCast(p));
|
||||
return null;
|
||||
}
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||
const old_payload = block.size - header_size;
|
||||
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||
rawFree(@ptrCast(p));
|
||||
return @ptrCast(np);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||
// In-place iff the new payload still fits the current block.
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||
return new_len + header_size <= block.size;
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
//! library/kernel/memory — the process's memory interface: the heap allocator, DMA-capable
|
||||
//! buffers, shared-memory regions, and the raw `mmap` grant they all sit on. One flat module
|
||||
//! (formerly runtime.heap / runtime.dma / runtime.shared_memory, plus the `mmap` wrappers that
|
||||
//! lived in the system.zig dumping ground). Its private files are heap.zig, dma.zig, and
|
||||
//! shared-memory.zig — imported only here, so the heap's state and C symbols exist once.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const heap = @import("heap.zig");
|
||||
const dma = @import("dma.zig");
|
||||
const shared = @import("shared-memory.zig");
|
||||
|
||||
// --- the heap: a std.mem.Allocator over a first-fit free list (C malloc/free are also
|
||||
// exported from heap.zig, compiled once here) ---
|
||||
pub const allocator = heap.allocator;
|
||||
|
||||
// --- the raw grant every allocation sits on ---
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable memory and
|
||||
/// return the base virtual address. On failure returns a value in the top page (`mmapFailed`).
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
/// Whether an `mmap` return value is an error (a wrapped -errno lands in the top page).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
// --- DMA-capable buffers: physically contiguous, pinned, uncacheable, physical address known ---
|
||||
pub const DmaRegion = dma.Region;
|
||||
pub const dma_coherent = dma.coherent;
|
||||
pub const dma_write_combining = dma.write_combining;
|
||||
pub const dma_below_4g = dma.below_4g;
|
||||
pub const dmaAlloc = dma.alloc;
|
||||
pub const dmaFree = dma.free;
|
||||
|
||||
// --- shared-memory regions: a capability handed to another process over an ipc_call send_cap ---
|
||||
pub const SharedRegion = shared.Region;
|
||||
pub const sharedCreate = shared.create;
|
||||
pub const sharedMap = shared.map;
|
||||
pub const sharedPhysical = shared.physical;
|
||||
@@ -0,0 +1,57 @@
|
||||
//! User-space shared memory: `shared_memory_create` / `shared_memory_map`. A process creates a shareable,
|
||||
//! zeroed, cacheable RAM region and gets back a pointer plus a **capability handle**; it
|
||||
//! passes that handle to another process as an `ipc_call` send_cap, and the receiver
|
||||
//! `shared_memory_map`s it to map the same physical pages. The kernel primitive under the display
|
||||
//! compositor↔native-driver and app↔compositor surface paths (docs/display-v2.md). The
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
}
|
||||
|
||||
/// A shared region: the `ptr` the CPU touches, and the `handle` (a capability) to hand to
|
||||
/// another process as an `ipc_call` send_cap.
|
||||
pub const Region = struct {
|
||||
ptr: [*]u8,
|
||||
handle: ipc.Handle,
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
|
||||
/// so this is a hand-written stub like `dma.alloc`.
|
||||
pub fn create(len: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: the capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.shared_memory_create)),
|
||||
[a0] "{rdi}" (len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .ptr = @ptrFromInt(rax), .handle = rdx, .len = len };
|
||||
}
|
||||
|
||||
/// Map the shared region named by a capability `handle` this process received (via an
|
||||
/// `ipc_call` send_cap) into its address space — the same physical pages the creator sees.
|
||||
/// Returns the pointer, or null on failure.
|
||||
pub fn map(handle: ipc.Handle) ?[*]u8 {
|
||||
const r = sc.systemCall1(.shared_memory_map, handle);
|
||||
if (failed(r)) return null;
|
||||
return @ptrFromInt(r);
|
||||
}
|
||||
|
||||
/// The guest-physical base of the shared region named by `handle` (which this process must
|
||||
/// hold a capability for). The region's frames are contiguous, so this single address plus
|
||||
/// the region length is all a device needs — e.g. a virtio-gpu driver programming an
|
||||
/// `attach_backing`. Returns null on failure.
|
||||
pub fn physical(handle: ipc.Handle) ?usize {
|
||||
const r = sc.systemCall1(.shared_memory_physical, handle);
|
||||
if (failed(r)) return null;
|
||||
return r;
|
||||
}
|
||||
Reference in New Issue
Block a user