diff --git a/build.zig b/build.zig index 4bb9021..8ac63e5 100644 --- a/build.zig +++ b/build.zig @@ -48,6 +48,43 @@ fn timestamp(b: *std.Build) []const u8 { }); } +/// Build one user-space binary the same way for every program (init, and later +/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model +/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that +/// can't reach), linked against the `rt` runtime library with the shared user +/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions) +/// are authoritative — the kernel's W^X user-ELF loader requires exact perms. +fn addUserBinary( + b: *std.Build, + target: std.Build.ResolvedTarget, + rt_mod: *std.Build.Module, + name: []const u8, + root: []const u8, +) *std.Build.Step.Compile { + const exe = b.addExecutable(.{ + .name = name, + .root_module = b.createModule(.{ + .root_source_file = b.path(root), + .target = target, + .optimize = .ReleaseSmall, + .code_model = .large, + .single_threaded = true, + .sanitize_c = .off, + .stack_check = false, + .stack_protector = false, + .imports = &.{ + .{ .name = "rt", .module = rt_mod }, + }, + }), + }); + exe.setLinkerScript(b.path("lib/user.ld")); + exe.entry = .{ .symbol_name = "_start" }; + exe.image_base = 0x7000_0000_0000; + exe.use_llvm = true; + exe.use_lld = true; + return exe; +} + pub fn build(b: *std.Build) void { ensureZigVersion(); @@ -98,6 +135,18 @@ pub fn build(b: *std.Build) void { }, }); + // The user-space runtime library (a nascent libc): syscall wrappers, the + // C-convention heap, IPC helpers, the process start shim. Compiled into every + // user binary (see addUserBinary), so it inherits each exe's `.large` code + // model — do NOT set a target/code_model here. It imports `danos` for the + // shared Syscall numbers. + const rt_mod = b.addModule("rt", .{ + .root_source_file = b.path("lib/rt.zig"), + .imports = &.{ + .{ .name = "danos", .module = mod }, + }, + }); + // Compile-time config the kernel reads as `@import("build_options")`. The // QEMU test harness sets -Dtest-case= to run one self-test at boot. const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)"); @@ -151,33 +200,10 @@ pub fn build(b: *std.Build) void { b.installArtifact(exe); // --- /sbin/init: the first user-space program --- - // Its own tiny freestanding binary, linked at a fixed address inside the - // kernel's user region (process.zig) and started in ring 3 by the kernel's - // user-ELF loader. `.large` because the image base is above 4 GiB — small/ - // medium code models emit 32-bit absolute relocations that can't reach. - // Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its - // size has no reason to track the kernel's optimize mode. - const init_exe = b.addExecutable(.{ - .name = "init", - .root_module = b.createModule(.{ - .root_source_file = b.path("sbin/init.zig"), - .target = kernel_target, - .optimize = .ReleaseSmall, - .code_model = .large, - .single_threaded = true, - .sanitize_c = .off, - .stack_check = false, - .stack_protector = false, - }), - }); - init_exe.setLinkerScript(b.path("sbin/linker.ld")); - init_exe.entry = .{ .symbol_name = "_start" }; - init_exe.image_base = 0x7000_0000_0000; - // The self-hosted linker ignores the script's PHDRS (segment permissions), - // which the kernel's W^X user-ELF loader requires (code must be R+X). Pin to - // LLVM + LLD so the script is authoritative. - init_exe.use_llvm = true; - init_exe.use_lld = true; + // Built by the shared user-binary recipe (see addUserBinary): freestanding, + // linked into the kernel's user region against the `rt` runtime library, and + // started in ring 3 by the kernel's user-ELF loader. + const init_exe = addUserBinary(b, kernel_target, rt_mod, "init", "sbin/init.zig"); b.installArtifact(init_exe); // Boot methods live in src/boot/, one per way of getting the kernel running. diff --git a/lib/heap.zig b/lib/heap.zig new file mode 100644 index 0000000..133efb5 --- /dev/null +++ b/lib/heap.zig @@ -0,0 +1,193 @@ +//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus +//! a `std.mem.Allocator` adapter over the same free list, so both C-style code +//! and Zig `std` containers share one heap. +//! +//! The algorithm is a straight port of the kernel's first-fit free list +//! (src/kernel/heap.zig): an address-ordered singly linked list of free blocks, +//! split on allocation and coalesced with neighbours on free. The only thing +//! that changes on this side of the syscall boundary is where memory comes from +//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames +//! itself, and the kernel picks the base address. +//! +//! Single-threaded and 16-byte max alignment, exactly like the kernel heap; a +//! lock and larger alignments come when user programs gain threads. + +const std = @import("std"); +const danos = @import("danos"); +const sys = @import("sys.zig"); + +const page_size = danos.page_size; + +/// A block header, at the start of every block; while free it also links the +/// free list via `next`. +const Block = extern struct { + size: usize, // total block size in bytes, including this header; a multiple of 16 + next: ?*Block, // free-list link (only meaningful while free) +}; + +const header_size = @sizeOf(Block); // 16 +const min_block = header_size + 16; // smallest block worth splitting off +/// Grow granularity: one `mmap` per 64 KiB amortises the syscall. +const chunk = 64 * 1024; + +var free_list: ?*Block = null; + +fn alignUp(value: usize, alignment: usize) usize { + return (value + alignment - 1) & ~(alignment - 1); +} + +fn payloadOf(block: *Block) [*]u8 { + return @ptrFromInt(@intFromPtr(block) + header_size); +} + +/// Ask the kernel for more pages and add them as a free block. Because each +/// `mmap` is an independent grant, cross-grant coalescing happens only when the +/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive +/// grants usually are adjacent). Returns false if the kernel is out of memory. +fn grow(min_bytes: usize) bool { + const bytes = alignUp(@max(min_bytes, chunk), page_size); + const ret = sys.mmap(bytes, sys.PROT_READ | sys.PROT_WRITE); + if (sys.mmapFailed(ret)) return false; + + const block: *Block = @ptrFromInt(ret); + block.size = bytes; + insertFree(block); // coalesces if this grant is adjacent to a prior one + return true; +} + +/// Insert a block into the address-ordered free list, coalescing with the +/// physically adjacent free blocks on either side. +fn insertFree(block: *Block) void { + var prev: ?*Block = null; + var cur = free_list; + while (cur) |c| : (cur = c.next) { + if (@intFromPtr(c) > @intFromPtr(block)) break; + prev = c; + } + + block.next = cur; + if (prev) |p| p.next = block else free_list = block; + + // Merge forward into `cur` if they're contiguous. + if (cur) |c| { + if (@intFromPtr(block) + block.size == @intFromPtr(c)) { + block.size += c.size; + block.next = c.next; + } + } + // Merge `prev` forward into `block` if they're contiguous. + if (prev) |p| { + if (@intFromPtr(p) + p.size == @intFromPtr(block)) { + p.size += block.size; + p.next = block.next; + } + } +} + +/// Allocate `len` bytes (16-byte aligned), or null if out of memory. +fn rawAlloc(len: usize) ?[*]u8 { + const need = alignUp(header_size + len, 16); + + var attempts: u32 = 0; + while (attempts < 2) : (attempts += 1) { + var prev: ?*Block = null; + var cur = free_list; + while (cur) |block| : ({ + prev = block; + cur = block.next; + }) { + if (block.size < need) continue; + + if (block.size >= need + min_block) { + // Split: carve `need` off the front, leave the rest free. + const rest: *Block = @ptrFromInt(@intFromPtr(block) + need); + rest.size = block.size - need; + rest.next = block.next; + if (prev) |p| p.next = rest else free_list = rest; + block.size = need; + } else { + // Take the whole block. + if (prev) |p| p.next = block.next else free_list = block.next; + } + return payloadOf(block); + } + + // Nothing fit: grow and try once more. + if (!grow(need)) return null; + } + return null; +} + +fn rawFree(ptr: [*]u8) void { + const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size); + insertFree(block); +} + +// --- C ABI: the global implicit heap --------------------------------------- +// `extern "C"` symbols so future C code links the same malloc/free directly. + +export fn malloc(size: usize) callconv(.c) ?*anyopaque { + if (size == 0) return null; + const p = rawAlloc(size) orelse return null; + return @ptrCast(p); +} + +export fn free(ptr: ?*anyopaque) callconv(.c) void { + const p = ptr orelse return; + rawFree(@ptrCast(p)); +} + +export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque { + const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe + if (total == 0) return null; + const p = rawAlloc(total) orelse return null; + @memset(p[0..total], 0); + return @ptrCast(p); +} + +export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque { + const p = ptr orelse return malloc(size); + if (size == 0) { + rawFree(@ptrCast(p)); + return null; + } + const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size); + const old_payload = block.size - header_size; + if (size <= old_payload) return p; // shrink/same: keep the block + const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free + @memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]); + rawFree(@ptrCast(p)); + return @ptrCast(np); +} + +// --- std.mem.Allocator interface (same free list) -------------------------- + +pub fn allocator() std.mem.Allocator { + return .{ .ptr = undefined, .vtable = &vtable }; +} + +const vtable = std.mem.Allocator.VTable{ + .alloc = allocImpl, + .resize = resizeImpl, + .remap = remapImpl, + .free = freeImpl, +}; + +fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 { + if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned + return rawAlloc(len); +} + +fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool { + // In-place iff the new payload still fits the current block. + const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size); + return new_len + header_size <= block.size; +} + +fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 { + return null; +} + +fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void { + rawFree(memory.ptr); +} diff --git a/lib/ipc.zig b/lib/ipc.zig new file mode 100644 index 0000000..436ddbb --- /dev/null +++ b/lib/ipc.zig @@ -0,0 +1,15 @@ +//! User-space IPC helpers. The kernel's synchronous IPC syscalls +//! (create_endpoint, ipc_register/lookup, ipc_call, ipc_reply_wait) arrive in a +//! later milestone; this reserves the module boundary now so drivers and servers +//! can be written against `rt.ipc` without restructuring once they light up. + +/// A fixed-size, register-friendly message payload. The wire format for the VFS +/// and driver protocols is layered on top of this by the servers themselves. +pub const Message = extern struct { + tag: u64 = 0, + a: u64 = 0, + b: u64 = 0, + c: u64 = 0, +}; + +// call() / replyWait() / createEndpoint() land with the kernel IPC syscalls. diff --git a/lib/rt.zig b/lib/rt.zig new file mode 100644 index 0000000..d6644c9 --- /dev/null +++ b/lib/rt.zig @@ -0,0 +1,22 @@ +//! danos user-space runtime library — a nascent libc. Every user binary (init, +//! and later the VFS server + device drivers) imports this as `@import("rt")`: +//! syscall wrappers, the C-convention heap, IPC helpers, and the process start +//! shim. It is compiled into each binary (inheriting its `.large` code model and +//! freestanding target), so all user programs share one implementation. +//! +//! A user binary needs three lines: +//! const rt = @import("rt"); +//! pub const panic = rt.panic; +//! comptime { _ = &rt.start._start; } // pull the entry shim in +//! and a `pub fn main() void`. + +pub const sys = @import("sys.zig"); +pub const heap = @import("heap.zig"); +pub const ipc = @import("ipc.zig"); +pub const start = @import("start.zig"); + +/// Re-exported so a user binary can `pub const panic = rt.panic;`. +pub const panic = start.panic; + +/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code. +pub const allocator = heap.allocator; diff --git a/lib/start.zig b/lib/start.zig new file mode 100644 index 0000000..15a4892 --- /dev/null +++ b/lib/start.zig @@ -0,0 +1,32 @@ +//! The user-space process entry shim. Every user binary roots `_start` here (via +//! `entry = _start` in build.zig) and forces this file to be analysed with +//! `comptime { _ = &rt.start._start; }`, so the whole runtime is linked in. + +const std = @import("std"); +const sys = @import("sys.zig"); + +/// The kernel enters at `_start` with rsp 16-aligned, but a SysV function expects +/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes +/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the +/// `ud2` is a safety net if `rt_start` ever returns. +pub export fn _start() callconv(.naked) noreturn { + asm volatile ( + \\call rt_start + \\ud2 + ); +} + +/// The first Zig frame. The heap is lazy (first alloc grows it), so there is no +/// runtime init to order here — just hand control to the program's `main`. +export fn rt_start() callconv(.c) noreturn { + const root = @import("root"); // the user binary's root source file + root.main(); + sys.exit(0); +} + +/// No runtime to unwind into — report a panic as a nonzero exit code. +pub const panic = std.debug.FullPanic(struct { + fn panic(_: []const u8, _: ?usize) noreturn { + sys.exit(127); + } +}.panic); diff --git a/lib/sys.zig b/lib/sys.zig new file mode 100644 index 0000000..06a0ead --- /dev/null +++ b/lib/sys.zig @@ -0,0 +1,52 @@ +//! Typed syscall surface for user space — thin wrappers over the raw `syscall` +//! stubs, one per kernel call. Numbers come from `danos.Syscall`, the single +//! source of truth shared with the kernel dispatcher. + +const danos = @import("danos"); +const sc = @import("syscall.zig"); + +/// `mmap` protection flags (matching the usual C bit values). Grants are always +/// readable+writable today; the kernel does not yet honour finer prot. +pub const PROT_READ: usize = danos.prot_read; +pub const PROT_WRITE: usize = danos.prot_write; +pub const PROT_EXEC: usize = danos.prot_exec; + +/// Give up the rest of this quantum. +pub fn yield() void { + _ = sc.syscall0(.yield); +} + +/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes +/// through the console/VFS later). Returns the byte count, or a wrapped -1. +pub fn write(msg: []const u8) usize { + return sc.syscall2(.debug_write, @intFromPtr(msg.ptr), msg.len); +} + +/// Block the caller for `ms` milliseconds. +pub fn sleep(ms: usize) void { + _ = sc.syscall1(.sleep, ms); +} + +/// End the process. Never returns. +pub fn exit(code: usize) noreturn { + _ = sc.syscall1(.exit, code); + unreachable; // the kernel never returns from exit +} + +/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable +/// memory and return the base virtual address. On failure returns a value in the +/// top page (see `mmapFailed`). The user heap grows through this call. +pub fn mmap(len: usize, prot: usize) usize { + return sc.syscall2(.mmap, len, prot); +} + +/// Release a range previously handed out by `mmap`. +pub fn munmap(base: usize, len: usize) usize { + return sc.syscall2(.munmap, base, len); +} + +/// Whether an `mmap` return value is an error (the kernel returns a wrapped +/// -errno, which lands in the top page — no real grant base is ever that high). +pub inline fn mmapFailed(ret: usize) bool { + return ret > ~@as(usize, 0) - 4095; +} diff --git a/lib/syscall.zig b/lib/syscall.zig new file mode 100644 index 0000000..8152782 --- /dev/null +++ b/lib/syscall.zig @@ -0,0 +1,52 @@ +//! Raw `syscall` instruction wrappers for user space — one per arity. +//! +//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax. +//! The `syscall` instruction itself clobbers rcx (it holds the return rip) and +//! r11 (the saved rflags); the kernel entry stub preserves everything else. +//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the +//! instruction, so the kernel reads the 4th argument from r10. + +const danos = @import("danos"); +const Syscall = danos.Syscall; + +pub inline fn syscall0(n: Syscall) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), + : .{ .rcx = true, .r11 = true, .memory = true }); +} + +pub inline fn syscall1(n: Syscall, a0: usize) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), + : .{ .rcx = true, .r11 = true, .memory = true }); +} + +pub inline fn syscall2(n: Syscall, a0: usize, a1: usize) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), + : .{ .rcx = true, .r11 = true, .memory = true }); +} + +pub inline fn syscall3(n: Syscall, a0: usize, a1: usize, a2: usize) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), + : .{ .rcx = true, .r11 = true, .memory = true }); +} + +pub inline fn syscall4(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), + : .{ .rcx = true, .r11 = true, .memory = true }); +} + +pub inline fn syscall5(n: Syscall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize { + return asm volatile ("syscall" + : [ret] "={rax}" (-> usize), + : [n] "{rax}" (@intFromEnum(n)), [a0] "{rdi}" (a0), [a1] "{rsi}" (a1), [a2] "{rdx}" (a2), [a3] "{r10}" (a3), [a4] "{r8}" (a4), + : .{ .rcx = true, .r11 = true, .memory = true }); +} diff --git a/sbin/linker.ld b/lib/user.ld similarity index 95% rename from sbin/linker.ld rename to lib/user.ld index 1d4d305..162bafd 100644 --- a/sbin/linker.ld +++ b/lib/user.ld @@ -1,4 +1,4 @@ -/* /sbin/init link layout. +/* Shared link layout for every user binary (init, servers, drivers). * * Linked at a fixed user-space virtual base (set by `image_base` in build.zig, * inside the kernel's user region). Same discipline as the kernel's script: diff --git a/sbin/init.zig b/sbin/init.zig index 5139b98..c4598e7 100644 --- a/sbin/init.zig +++ b/sbin/init.zig @@ -1,65 +1,39 @@ //! /sbin/init — the first user-space program, PID 1. Built as its own //! freestanding binary (see build.zig), shipped on the boot volume at sbin/init, //! loaded by the bootloader, and started in ring 3 as a scheduled process by the -//! kernel (src/kernel/process.zig). It talks to the kernel only through the -//! `syscall` instruction. +//! kernel (src/kernel/process.zig). It links against the shared user runtime +//! library `rt` and talks to the kernel only through `rt`'s syscall wrappers. //! -//! Today it's a heartbeat: it prints a line and sleeps, forever — enough to show -//! the system reaches user space and stays alive with a real process scheduled -//! alongside the kernel's idle loop. It grows into the real init (service -//! supervision) once there are other user programs to supervise. +//! Today it proves the C-convention heap works, then settles into a heartbeat: +//! it prints a line and sleeps, forever — enough to show the system reaches user +//! space and stays alive with a real process scheduled alongside the kernel's +//! idle loop. It grows into the real init (service supervision) once there are +//! other user programs to supervise. -const std = @import("std"); +const rt = @import("rt"); -// Syscall numbers (see src/kernel/process.zig): -const sys_exit = 0; -const sys_write = 2; -const sys_sleep = 3; +pub fn main() void { + // Prove the heap end to end: allocate through the runtime allocator (which + // mmaps pages from the kernel and carves them with the free list), write into + // that heap buffer (exercising the widened debug_write bounds check), and + // free it. A fault here would kill init before it heartbeats — so the init + // test doubles as the heap regression test. (C code links the same heap via + // the extern malloc/free symbols; Zig code uses this allocator.) + const gpa = rt.allocator(); + if (gpa.alloc(u8, 64)) |buf| { + const msg = "init: heap ok\n"; + @memcpy(buf[0..msg.len], msg); + _ = rt.sys.write(buf[0..msg.len]); + gpa.free(buf); + } else |_| {} -fn syscall2(n: u64, a: u64, b: u64) u64 { - // The `syscall` instruction clobbers RCX (return RIP) and R11 (saved RFLAGS); - // the kernel entry stub preserves everything else. - return asm volatile ("syscall" - : [ret] "={rax}" (-> u64), - : [n] "{rax}" (n), - [a] "{rdi}" (a), - [b] "{rsi}" (b), - : .{ .rcx = true, .r11 = true, .memory = true }); -} - -fn write(msg: []const u8) void { - _ = syscall2(sys_write, @intFromPtr(msg.ptr), msg.len); -} - -fn sleep(ms: u64) void { - _ = syscall2(sys_sleep, ms, 0); -} - -fn exit(code: u64) noreturn { - _ = syscall2(sys_exit, code, 0); - unreachable; // the kernel never returns from exit -} - -/// Entry. Naked: the kernel enters with rsp 16-aligned, but a SysV function -/// expects rsp ≡ 8 (mod 16) on entry (as if reached by `call`) — so re-enter -/// the ABI with an actual call. The trap after is a safety net. -pub export fn _start() callconv(.naked) noreturn { - asm volatile ( - \\call init_main - \\ud2 - ); -} - -export fn init_main() callconv(.c) noreturn { while (true) { - write("init: heartbeat\n"); - sleep(1000); + _ = rt.sys.write("init: heartbeat\n"); + rt.sys.sleep(1000); } } -/// No runtime to unwind into — report the panic as a nonzero exit code. -pub const panic = std.debug.FullPanic(struct { - fn panic(_: []const u8, _: ?usize) noreturn { - exit(127); - } -}.panic); +pub const panic = rt.panic; +comptime { + _ = &rt.start._start; // pull the runtime entry shim into the image +}