M6: user runtime library rt + C-convention heap

Add lib/ — the shared user-space runtime every user binary links against
(init now, servers/drivers later): syscall wrappers, the heap, IPC stub,
and the process start shim.

- lib/heap.zig: the kernel first-fit free-list ported to user space, grown
  via the mmap syscall instead of pmm+mapPage. Dual API over one global free
  list: extern "C" malloc/free/calloc/realloc (C ABI for future C code) and a
  std.mem.Allocator adapter (with in-place resize) for Zig std containers.
- lib/syscall.zig + sys.zig: raw syscall0..5 (arg3 in r10) and typed
  yield/write/sleep/exit/mmap/munmap over danos.Syscall.
- lib/start.zig: naked _start -> rt_start -> root.main() (SysV realign via
  call), panic -> exit(127).
- lib/user.ld: the user link script, moved from sbin/linker.ld (shared by all
  user binaries).
- build.zig: register the `rt` module; add an addUserBinary() helper that is
  the one recipe for every user binary (freestanding, .large, use_lld,
  user.ld, image_base), replacing the bespoke init block.
- sbin/init.zig: migrated onto rt; drops its hand-rolled syscall2/shims. Now
  proves the heap (alloc -> write from a heap pointer -> free) before the
  heartbeat loop. Serial shows "init: heap ok". Suite 28/28.
This commit is contained in:
Daniel Samson
2026-07-09 07:08:32 +01:00
parent 9316f9f1c3
commit 0a8c81b17a
9 changed files with 448 additions and 82 deletions
+53 -27
View File
@@ -48,6 +48,43 @@ fn timestamp(b: *std.Build) []const u8 {
});
}
/// Build one user-space binary the same way for every program (init, and later
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
/// can't reach), linked against the `rt` runtime library with the shared user
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
fn addUserBinary(
b: *std.Build,
target: std.Build.ResolvedTarget,
rt_mod: *std.Build.Module,
name: []const u8,
root: []const u8,
) *std.Build.Step.Compile {
const exe = b.addExecutable(.{
.name = name,
.root_module = b.createModule(.{
.root_source_file = b.path(root),
.target = target,
.optimize = .ReleaseSmall,
.code_model = .large,
.single_threaded = true,
.sanitize_c = .off,
.stack_check = false,
.stack_protector = false,
.imports = &.{
.{ .name = "rt", .module = rt_mod },
},
}),
});
exe.setLinkerScript(b.path("lib/user.ld"));
exe.entry = .{ .symbol_name = "_start" };
exe.image_base = 0x7000_0000_0000;
exe.use_llvm = true;
exe.use_lld = true;
return exe;
}
pub fn build(b: *std.Build) void {
ensureZigVersion();
@@ -98,6 +135,18 @@ pub fn build(b: *std.Build) void {
},
});
// The user-space runtime library (a nascent libc): syscall wrappers, the
// C-convention heap, IPC helpers, the process start shim. Compiled into every
// user binary (see addUserBinary), so it inherits each exe's `.large` code
// model — do NOT set a target/code_model here. It imports `danos` for the
// shared Syscall numbers.
const rt_mod = b.addModule("rt", .{
.root_source_file = b.path("lib/rt.zig"),
.imports = &.{
.{ .name = "danos", .module = mod },
},
});
// Compile-time config the kernel reads as `@import("build_options")`. The
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
@@ -151,33 +200,10 @@ pub fn build(b: *std.Build) void {
b.installArtifact(exe);
// --- /sbin/init: the first user-space program ---
// Its own tiny freestanding binary, linked at a fixed address inside the
// kernel's user region (process.zig) and started in ring 3 by the kernel's
// user-ELF loader. `.large` because the image base is above 4 GiB — small/
// medium code models emit 32-bit absolute relocations that can't reach.
// Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its
// size has no reason to track the kernel's optimize mode.
const init_exe = b.addExecutable(.{
.name = "init",
.root_module = b.createModule(.{
.root_source_file = b.path("sbin/init.zig"),
.target = kernel_target,
.optimize = .ReleaseSmall,
.code_model = .large,
.single_threaded = true,
.sanitize_c = .off,
.stack_check = false,
.stack_protector = false,
}),
});
init_exe.setLinkerScript(b.path("sbin/linker.ld"));
init_exe.entry = .{ .symbol_name = "_start" };
init_exe.image_base = 0x7000_0000_0000;
// The self-hosted linker ignores the script's PHDRS (segment permissions),
// which the kernel's W^X user-ELF loader requires (code must be R+X). Pin to
// LLVM + LLD so the script is authoritative.
init_exe.use_llvm = true;
init_exe.use_lld = true;
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
// linked into the kernel's user region against the `rt` runtime library, and
// started in ring 3 by the kernel's user-ELF loader.
const init_exe = addUserBinary(b, kernel_target, rt_mod, "init", "sbin/init.zig");
b.installArtifact(init_exe);
// Boot methods live in src/boot/, one per way of getting the kernel running.
+193
View File
@@ -0,0 +1,193 @@
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
//! and Zig `std` containers share one heap.
//!
//! The algorithm is a straight port of the kernel's first-fit free list
//! (src/kernel/heap.zig): an address-ordered singly linked list of free blocks,
//! split on allocation and coalesced with neighbours on free. The only thing
//! that changes on this side of the syscall boundary is where memory comes from
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
//! itself, and the kernel picks the base address.
//!
//! Single-threaded and 16-byte max alignment, exactly like the kernel heap; a
//! lock and larger alignments come when user programs gain threads.
const std = @import("std");
const danos = @import("danos");
const sys = @import("sys.zig");
const page_size = danos.page_size;
/// A block header, at the start of every block; while free it also links the
/// free list via `next`.
const Block = extern struct {
size: usize, // total block size in bytes, including this header; a multiple of 16
next: ?*Block, // free-list link (only meaningful while free)
};
const header_size = @sizeOf(Block); // 16
const min_block = header_size + 16; // smallest block worth splitting off
/// Grow granularity: one `mmap` per 64 KiB amortises the syscall.
const chunk = 64 * 1024;
var free_list: ?*Block = null;
fn alignUp(value: usize, alignment: usize) usize {
return (value + alignment - 1) & ~(alignment - 1);
}
fn payloadOf(block: *Block) [*]u8 {
return @ptrFromInt(@intFromPtr(block) + header_size);
}
/// Ask the kernel for more pages and add them as a free block. Because each
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
/// grants usually are adjacent). Returns false if the kernel is out of memory.
fn grow(min_bytes: usize) bool {
const bytes = alignUp(@max(min_bytes, chunk), page_size);
const ret = sys.mmap(bytes, sys.PROT_READ | sys.PROT_WRITE);
if (sys.mmapFailed(ret)) return false;
const block: *Block = @ptrFromInt(ret);
block.size = bytes;
insertFree(block); // coalesces if this grant is adjacent to a prior one
return true;
}
/// Insert a block into the address-ordered free list, coalescing with the
/// physically adjacent free blocks on either side.
fn insertFree(block: *Block) void {
var prev: ?*Block = null;
var cur = free_list;
while (cur) |c| : (cur = c.next) {
if (@intFromPtr(c) > @intFromPtr(block)) break;
prev = c;
}
block.next = cur;
if (prev) |p| p.next = block else free_list = block;
// Merge forward into `cur` if they're contiguous.
if (cur) |c| {
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
block.size += c.size;
block.next = c.next;
}
}
// Merge `prev` forward into `block` if they're contiguous.
if (prev) |p| {
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
p.size += block.size;
p.next = block.next;
}
}
}
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
fn rawAlloc(len: usize) ?[*]u8 {
const need = alignUp(header_size + len, 16);
var attempts: u32 = 0;
while (attempts < 2) : (attempts += 1) {
var prev: ?*Block = null;
var cur = free_list;
while (cur) |block| : ({
prev = block;
cur = block.next;
}) {
if (block.size < need) continue;
if (block.size >= need + min_block) {
// Split: carve `need` off the front, leave the rest free.
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
rest.size = block.size - need;
rest.next = block.next;
if (prev) |p| p.next = rest else free_list = rest;
block.size = need;
} else {
// Take the whole block.
if (prev) |p| p.next = block.next else free_list = block.next;
}
return payloadOf(block);
}
// Nothing fit: grow and try once more.
if (!grow(need)) return null;
}
return null;
}
fn rawFree(ptr: [*]u8) void {
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
insertFree(block);
}
// --- C ABI: the global implicit heap ---------------------------------------
// `extern "C"` symbols so future C code links the same malloc/free directly.
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
if (size == 0) return null;
const p = rawAlloc(size) orelse return null;
return @ptrCast(p);
}
export fn free(ptr: ?*anyopaque) callconv(.c) void {
const p = ptr orelse return;
rawFree(@ptrCast(p));
}
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
if (total == 0) return null;
const p = rawAlloc(total) orelse return null;
@memset(p[0..total], 0);
return @ptrCast(p);
}
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
const p = ptr orelse return malloc(size);
if (size == 0) {
rawFree(@ptrCast(p));
return null;
}
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
const old_payload = block.size - header_size;
if (size <= old_payload) return p; // shrink/same: keep the block
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
rawFree(@ptrCast(p));
return @ptrCast(np);
}
// --- std.mem.Allocator interface (same free list) --------------------------
pub fn allocator() std.mem.Allocator {
return .{ .ptr = undefined, .vtable = &vtable };
}
const vtable = std.mem.Allocator.VTable{
.alloc = allocImpl,
.resize = resizeImpl,
.remap = remapImpl,
.free = freeImpl,
};
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
return rawAlloc(len);
}
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
// In-place iff the new payload still fits the current block.
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
return new_len + header_size <= block.size;
}
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
return null;
}
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
rawFree(memory.ptr);
}
+15
View File
@@ -0,0 +1,15 @@
//! User-space IPC helpers. The kernel's synchronous IPC syscalls
//! (create_endpoint, ipc_register/lookup, ipc_call, ipc_reply_wait) arrive in a
//! later milestone; this reserves the module boundary now so drivers and servers
//! can be written against `rt.ipc` without restructuring once they light up.
/// A fixed-size, register-friendly message payload. The wire format for the VFS
/// and driver protocols is layered on top of this by the servers themselves.
pub const Message = extern struct {
tag: u64 = 0,
a: u64 = 0,
b: u64 = 0,
c: u64 = 0,
};
// call() / replyWait() / createEndpoint() land with the kernel IPC syscalls.
+22
View File
@@ -0,0 +1,22 @@
//! danos user-space runtime library — a nascent libc. Every user binary (init,
//! and later the VFS server + device drivers) imports this as `@import("rt")`:
//! syscall wrappers, the C-convention heap, IPC helpers, and the process start
//! shim. It is compiled into each binary (inheriting its `.large` code model and
//! freestanding target), so all user programs share one implementation.
//!
//! A user binary needs three lines:
//! const rt = @import("rt");
//! pub const panic = rt.panic;
//! comptime { _ = &rt.start._start; } // pull the entry shim in
//! and a `pub fn main() void`.
pub const sys = @import("sys.zig");
pub const heap = @import("heap.zig");
pub const ipc = @import("ipc.zig");
pub const start = @import("start.zig");
/// Re-exported so a user binary can `pub const panic = rt.panic;`.
pub const panic = start.panic;
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
pub const allocator = heap.allocator;
+32
View File
@@ -0,0 +1,32 @@
//! The user-space process entry shim. Every user binary roots `_start` here (via
//! `entry = _start` in build.zig) and forces this file to be analysed with
//! `comptime { _ = &rt.start._start; }`, so the whole runtime is linked in.
const std = @import("std");
const sys = @import("sys.zig");
/// The kernel enters at `_start` with rsp 16-aligned, but a SysV function expects
/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes
/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the
/// `ud2` is a safety net if `rt_start` ever returns.
pub export fn _start() callconv(.naked) noreturn {
asm volatile (
\\call rt_start
\\ud2
);
}
/// The first Zig frame. The heap is lazy (first alloc grows it), so there is no
/// runtime init to order here — just hand control to the program's `main`.
export fn rt_start() callconv(.c) noreturn {
const root = @import("root"); // the user binary's root source file
root.main();
sys.exit(0);
}
/// No runtime to unwind into — report a panic as a nonzero exit code.
pub const panic = std.debug.FullPanic(struct {
fn panic(_: []const u8, _: ?usize) noreturn {
sys.exit(127);
}
}.panic);
+52
View File
@@ -0,0 +1,52 @@
//! Typed syscall surface for user space — thin wrappers over the raw `syscall`
//! stubs, one per kernel call. Numbers come from `danos.Syscall`, the single
//! source of truth shared with the kernel dispatcher.
const danos = @import("danos");
const sc = @import("syscall.zig");
/// `mmap` protection flags (matching the usual C bit values). Grants are always
/// readable+writable today; the kernel does not yet honour finer prot.
pub const PROT_READ: usize = danos.prot_read;
pub const PROT_WRITE: usize = danos.prot_write;
pub const PROT_EXEC: usize = danos.prot_exec;
/// Give up the rest of this quantum.
pub fn yield() void {
_ = sc.syscall0(.yield);
}
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
pub fn write(msg: []const u8) usize {
return sc.syscall2(.debug_write, @intFromPtr(msg.ptr), msg.len);
}
/// Block the caller for `ms` milliseconds.
pub fn sleep(ms: usize) void {
_ = sc.syscall1(.sleep, ms);
}
/// End the process. Never returns.
pub fn exit(code: usize) noreturn {
_ = sc.syscall1(.exit, code);
unreachable; // the kernel never returns from exit
}
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
/// memory and return the base virtual address. On failure returns a value in the
/// top page (see `mmapFailed`). The user heap grows through this call.
pub fn mmap(len: usize, prot: usize) usize {
return sc.syscall2(.mmap, len, prot);
}
/// Release a range previously handed out by `mmap`.
pub fn munmap(base: usize, len: usize) usize {
return sc.syscall2(.munmap, base, len);
}
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
/// -errno, which lands in the top page — no real grant base is ever that high).
pub inline fn mmapFailed(ret: usize) bool {
return ret > ~@as(usize, 0) - 4095;
}
+52
View File
@@ -0,0 +1,52 @@
//! Raw `syscall` instruction wrappers for user space — one per arity.
//!
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
//! The `syscall` instruction itself clobbers rcx (it holds the return rip) and
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
//! instruction, so the kernel reads the 4th argument from r10.
const danos = @import("danos");
const Syscall = danos.Syscall;
pub inline fn syscall0(n: Syscall) usize {
return asm volatile ("syscall"
: [ret] "={rax}" (-> usize),
: [n] "{rax}" (@intFromEnum(n)),
: .{ .rcx = true, .r11 = true, .memory = true });
}
pub inline fn syscall1(n: Syscall, a0: usize) usize {
return asm volatile ("syscall"
: [ret] "={rax}" (-> usize),
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0),
: .{ .rcx = true, .r11 = true, .memory = true });
}
pub inline fn syscall2(n: Syscall, a0: usize, a1: usize) usize {
return asm volatile ("syscall"
: [ret] "={rax}" (-> usize),
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1),
: .{ .rcx = true, .r11 = true, .memory = true });
}
pub inline fn syscall3(n: Syscall, a0: usize, a1: usize, a2: usize) usize {
return asm volatile ("syscall"
: [ret] "={rax}" (-> usize),
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2),
: .{ .rcx = true, .r11 = true, .memory = true });
}
pub inline fn syscall4(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
return asm volatile ("syscall"
: [ret] "={rax}" (-> usize),
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3),
: .{ .rcx = true, .r11 = true, .memory = true });
}
pub inline fn syscall5(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
return asm volatile ("syscall"
: [ret] "={rax}" (-> usize),
: [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), [a4] "{r8}" (a4),
: .{ .rcx = true, .r11 = true, .memory = true });
}
+1 -1
View File
@@ -1,4 +1,4 @@
/* /sbin/init link layout.
/* Shared link layout for every user binary (init, servers, drivers).
*
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
* inside the kernel's user region). Same discipline as the kernel's script:
+27 -53
View File
@@ -1,65 +1,39 @@
//! /sbin/init — the first user-space program, PID 1. Built as its own
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
//! kernel (src/kernel/process.zig). It talks to the kernel only through the
//! `syscall` instruction.
//! kernel (src/kernel/process.zig). It links against the shared user runtime
//! library `rt` and talks to the kernel only through `rt`'s syscall wrappers.
//!
//! Today it's a heartbeat: it prints a line and sleeps, forever — enough to show
//! the system reaches user space and stays alive with a real process scheduled
//! alongside the kernel's idle loop. It grows into the real init (service
//! supervision) once there are other user programs to supervise.
//! Today it proves the C-convention heap works, then settles into a heartbeat:
//! it prints a line and sleeps, forever — enough to show the system reaches user
//! space and stays alive with a real process scheduled alongside the kernel's
//! idle loop. It grows into the real init (service supervision) once there are
//! other user programs to supervise.
const std = @import("std");
const rt = @import("rt");
// Syscall numbers (see src/kernel/process.zig):
const sys_exit = 0;
const sys_write = 2;
const sys_sleep = 3;
pub fn main() void {
// Prove the heap end to end: allocate through the runtime allocator (which
// mmaps pages from the kernel and carves them with the free list), write into
// that heap buffer (exercising the widened debug_write bounds check), and
// free it. A fault here would kill init before it heartbeats — so the init
// test doubles as the heap regression test. (C code links the same heap via
// the extern malloc/free symbols; Zig code uses this allocator.)
const gpa = rt.allocator();
if (gpa.alloc(u8, 64)) |buf| {
const msg = "init: heap ok\n";
@memcpy(buf[0..msg.len], msg);
_ = rt.sys.write(buf[0..msg.len]);
gpa.free(buf);
} else |_| {}
fn syscall2(n: u64, a: u64, b: u64) u64 {
// The `syscall` instruction clobbers RCX (return RIP) and R11 (saved RFLAGS);
// the kernel entry stub preserves everything else.
return asm volatile ("syscall"
: [ret] "={rax}" (-> u64),
: [n] "{rax}" (n),
[a] "{rdi}" (a),
[b] "{rsi}" (b),
: .{ .rcx = true, .r11 = true, .memory = true });
}
fn write(msg: []const u8) void {
_ = syscall2(sys_write, @intFromPtr(msg.ptr), msg.len);
}
fn sleep(ms: u64) void {
_ = syscall2(sys_sleep, ms, 0);
}
fn exit(code: u64) noreturn {
_ = syscall2(sys_exit, code, 0);
unreachable; // the kernel never returns from exit
}
/// Entry. Naked: the kernel enters with rsp 16-aligned, but a SysV function
/// expects rsp ≡ 8 (mod 16) on entry (as if reached by `call`) — so re-enter
/// the ABI with an actual call. The trap after is a safety net.
pub export fn _start() callconv(.naked) noreturn {
asm volatile (
\\call init_main
\\ud2
);
}
export fn init_main() callconv(.c) noreturn {
while (true) {
write("init: heartbeat\n");
sleep(1000);
_ = rt.sys.write("init: heartbeat\n");
rt.sys.sleep(1000);
}
}
/// No runtime to unwind into — report the panic as a nonzero exit code.
pub const panic = std.debug.FullPanic(struct {
fn panic(_: []const u8, _: ?usize) noreturn {
exit(127);
pub const panic = rt.panic;
comptime {
_ = &rt.start._start; // pull the runtime entry shim into the image
}
}.panic);