From 9316f9f1c31efc116f536b3677d3acb03a26a906 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Thu, 9 Jul 2026 06:55:09 +0100 Subject: [PATCH] M5: rename usermode->process, add yield + mmap/munmap syscalls Start the user-space driver track (VFS + IPC + heap). This lays the process/syscall foundation the runtime heap will grow on. - Rename usermode.zig -> process.zig; drop the retired hello/ping blob and its `user` test (subsumed by the real /sbin/init exerciser). Keep the isolation-proof pf blob and the user-pf test. - Add danos.Syscall as the single source of truth for syscall numbers, shared by the kernel dispatcher and (later) the user runtime lib. Dispatch on the enum. New calls: 1=yield, 4=mmap, 5=munmap. Widen debug_write's bounds check to the whole user low half so heap buffers are writable. - mmap grants zeroed RW+NX pages from a per-process bump arena (Task.heap_next, PML4[224] above image+stack); munmap frees the frames. Add paging.translateIn / unmapInto (+ arch.translate / unmapUserPageInto) as the primitives munmap and future cross-AS copies need. - New `usermem` test: grant three pages into a fresh AS, translate them, release via the munmap path, tear down, and assert no frames leak. Suite 28/28 (user -> usermem). --- build.zig | 2 +- docs/syscall.md | 2 +- sbin/init.zig | 4 +- src/kernel/arch/x86_64/cpu.zig | 13 ++ src/kernel/arch/x86_64/isr.s | 28 +--- src/kernel/arch/x86_64/paging.zig | 28 +++- src/kernel/main.zig | 6 +- src/kernel/{usermode.zig => process.zig} | 200 +++++++++++++++-------- src/kernel/scheduler.zig | 8 +- src/kernel/tests.zig | 112 +++++++++---- src/root.zig | 20 +++ test/qemu_test.py | 6 +- 12 files changed, 290 insertions(+), 139 deletions(-) rename src/kernel/{usermode.zig => process.zig} (60%) diff --git a/build.zig b/build.zig index f7d81dd..4bb9021 100644 --- a/build.zig +++ b/build.zig @@ -152,7 +152,7 @@ pub fn build(b: *std.Build) void { // --- /sbin/init: the first user-space program --- // Its own tiny freestanding binary, linked at a fixed address inside the - // kernel's user region (usermode.zig) and started in ring 3 by the kernel's + // kernel's user region (process.zig) and started in ring 3 by the kernel's // user-ELF loader. `.large` because the image base is above 4 GiB — small/ // medium code models emit 32-bit absolute relocations that can't reach. // Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its diff --git a/docs/syscall.md b/docs/syscall.md index 8873928..6b8b52d 100644 --- a/docs/syscall.md +++ b/docs/syscall.md @@ -6,7 +6,7 @@ System calls (syscalls) are the bridge between your programs and the operating s > `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt > dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The > current call set is still a placeholder — `0 = exit(code)`, `1 = ping`, -> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/usermode.zig`); the +> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/process.zig`); the > handler dispatches on whether the caller is a scheduled process (its own address > space) or a borrowed test thread. The microkernel set below (IPC_Call / > IPC_ReplyWait / Yield) replaces it once a second user server exists. diff --git a/sbin/init.zig b/sbin/init.zig index 801b0af..5139b98 100644 --- a/sbin/init.zig +++ b/sbin/init.zig @@ -1,7 +1,7 @@ //! /sbin/init — the first user-space program, PID 1. Built as its own //! freestanding binary (see build.zig), shipped on the boot volume at sbin/init, //! loaded by the bootloader, and started in ring 3 as a scheduled process by the -//! kernel (src/kernel/usermode.zig). It talks to the kernel only through the +//! kernel (src/kernel/process.zig). It talks to the kernel only through the //! `syscall` instruction. //! //! Today it's a heartbeat: it prints a line and sleeps, forever — enough to show @@ -11,7 +11,7 @@ const std = @import("std"); -// Syscall numbers (see src/kernel/usermode.zig): +// Syscall numbers (see src/kernel/process.zig): const sys_exit = 0; const sys_write = 2; const sys_sleep = 3; diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index d75af64..6d52006 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -190,6 +190,19 @@ pub fn unmapPage(virt: u64) void { paging.unmap(virt); } +/// Remove a page mapping from address space `root` (for munmap of user pages). +/// Clears the leaf entry only; freeing the underlying frame is the caller's job. +pub fn unmapUserPageInto(root: u64, virt: u64) void { + paging.unmapInto(root, virt); +} + +/// Resolve `virt` to its physical address in the address space rooted at `root` +/// (any address space, not just the live one), or null if unmapped. Used to find +/// the frame behind a user page for munmap, and for cross-address-space copies. +pub fn translate(root: u64, virt: u64) ?u64 { + return paging.translateIn(root, virt); +} + /// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps /// W^X: code read-only + executable, data writable + no-execute. pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void { diff --git a/src/kernel/arch/x86_64/isr.s b/src/kernel/arch/x86_64/isr.s index f4888d6..6de32fc 100644 --- a/src/kernel/arch/x86_64/isr.s +++ b/src/kernel/arch/x86_64/isr.s @@ -235,34 +235,14 @@ user_saved_rsp: .skip 8 .text -# --- user-mode test programs ------------------------------------------------- -# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and +# --- user-mode test program -------------------------------------------------- +# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and # entered via enter_user. Position-independent (immediates and short jumps only). # In .rodata: these bytes are data to the kernel — they only execute at CPL 3 -# from the user mapping. +# from the user mapping. (The old hello/ping blob was retired once /sbin/init +# became the real ring-3 exerciser; only the isolation proof remains.) .section .rodata -# The hello program: ping syscall (rax=1) with a value, a delay loop long enough -# for several timer ticks to land while at CPL 3 (proving interrupt-from-user + -# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety -# net in case exit ever returns. -.global user_prog_start -.global user_prog_end -user_prog_start: - mov $1, %rax - mov $0xC0DE, %rdi - int $0x80 - mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG -1: dec %rcx - jnz 1b - mov $1, %rax - mov $0xBEEF, %rdi - int $0x80 - mov $0, %rax - int $0x80 -2: jmp 2b -user_prog_end: - # The isolation-proof program: read a kernel-only page from ring 3. The LAPIC # lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor # page, so this must take a #PF with error code 0x5 (present | user) before any diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index 8b2a271..48dcd29 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -306,7 +306,15 @@ pub fn setExecutable(phys: u64) void { /// Remove a mapping and flush it from the TLB. pub fn unmap(virt: u64) void { - const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF]; + unmapInto(kernel_pml4, virt); +} + +/// Remove a mapping from the address space rooted at `pml4` (a process's own +/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE — +/// the intermediate tables and any frame the PTE pointed at are left to the +/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables). +pub fn unmapInto(pml4: u64, virt: u64) void { + const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF]; if (pml4e & present == 0) return; const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF]; if (pdpte & present == 0) return; @@ -316,6 +324,24 @@ pub fn unmap(virt: u64) void { invalidate(virt); } +/// Resolve a virtual address to a physical one in the address space rooted at +/// `pml4`, walking the tables through the physmap (CR3-independent — works for +/// any address space, not just the live one). Returns null if `virt` is not +/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page +/// case. The foundation for cross-address-space copies and for munmap (which +/// needs the frame behind a user vaddr to free it). +pub fn translateIn(pml4: u64, virt: u64) ?u64 { + const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF]; + if (pml4e & present == 0) return null; + const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF]; + if (pdpte & present == 0) return null; + const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF]; + if (pde & present == 0) return null; + const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF]; + if (pte & present == 0) return null; + return (pte & addr_mask) | (virt & (page_size - 1)); +} + fn invalidate(virt: u64) void { // invlpg needs its operand via a register-indirect memory reference that Zig // inline asm won't form directly, so stage the address in a register first. diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 5d539d6..c82a07c 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -7,7 +7,7 @@ const log = @import("log.zig"); const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const scheduler = @import("scheduler.zig"); -const usermode = @import("usermode.zig"); +const process = @import("process.zig"); const platform = @import("platform"); const tests = @import("tests.zig"); const build_options = @import("build_options"); @@ -228,7 +228,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Install the syscall handler (int 0x80 gate + syscall stub) once, before any // user code runs. - usermode.init(); + process.init(); // Register the current context as the first task before enabling preemption. scheduler.init(4); @@ -263,7 +263,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { if (boot_info.init_len != 0) { status("starting /sbin/init...\n"); const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; - usermode.spawnProcess(image, 4) catch |err| { + process.spawnProcess(image, 4) catch |err| { statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)}); }; } else { diff --git a/src/kernel/usermode.zig b/src/kernel/process.zig similarity index 60% rename from src/kernel/usermode.zig rename to src/kernel/process.zig index 9d5e639..140c4f5 100644 --- a/src/kernel/usermode.zig +++ b/src/kernel/process.zig @@ -1,10 +1,14 @@ -//! Ring-3 execution. Two entry points: -//! - `spawnProcess` loads a user ELF (`/sbin/init`) into a fresh address space -//! and schedules it as a real preemptive ring-3 process on its own page -//! tables. This is the production path. -//! - `run` executes a raw code blob (the user/user-pf test programs) on the +//! User-space processes: loading a user ELF and running it in ring 3. danos is a +//! microkernel, so this only ever loads *user* binaries — there is no kernel-space +//! loader; in-kernel code is linked into the kernel image, not loaded here. +//! +//! Two entry points: +//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers) +//! into a fresh address space and schedules it as a real preemptive ring-3 +//! process on its own page tables. This is the production path. +//! - `run` executes a raw code blob (the user-pf isolation test program) on the //! *current* kernel context via the borrowed-thread path — a minimal probe of -//! the ring-transition mechanisms, kept for the tests. +//! the ring-transition mechanisms, kept for that test. //! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program //! talks to the kernel only through the syscall instruction (or the int 0x80 //! gate). The shared handler is installed once by `init`. @@ -25,6 +29,7 @@ const sync = @import("sync.zig"); const log = @import("log.zig"); const page_size = danos.page_size; +const Syscall = danos.Syscall; /// User virtual addresses. PML4 index 224 — a user-exclusive region, far from /// the identity map (low indices) and the vmm test address (index 128), so @@ -33,100 +38,163 @@ const page_size = danos.page_size; pub const code_virt: u64 = 0x0000_7000_0000_0000; pub const stack_virt: u64 = 0x0000_7000_0020_0000; -// The hand-assembled user program blobs (isr.s, .rodata). -const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" }); -const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" }); +/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image +/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each +/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a +/// 1 GiB window is far more than any user heap needs today. +pub const heap_arena_base: u64 = 0x0000_7000_1000_0000; +pub const heap_arena_end: u64 = heap_arena_base + (1 << 30); + +/// End of the user (low) canonical half. Any legitimate user pointer is below it; +/// used to bound the addresses a syscall will dereference on the caller's behalf. +pub const user_half_end: u64 = 0x0000_8000_0000_0000; + +/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small +/// chunks, so this bound is generous; it also caps the frame scratch array below. +const max_mmap_pages = 256; + +// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe. const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" }); const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" }); -/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit. -pub fn helloBlob() []const u8 { - return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)]; -} - /// The isolation-proof program: reads a kernel-only page, must #PF. pub fn pfBlob() []const u8 { return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)]; } -/// What a ping syscall recorded — evidence for the test to assert on. -pub const Ping = struct { value: u64 = 0, from_user: bool = false, ticks: u64 = 0 }; -pub var pings = [_]Ping{.{}} ** 2; -pub var ping_count: usize = 0; - -/// What write syscalls produced (accumulated), and the exit syscall's code. +/// What debug_write syscalls produced (accumulated), and the exit syscall's code. pub var write_buf: [256]u8 = undefined; pub var write_len: usize = 0; pub var write_from_user: bool = false; pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests) pub var exit_code: u64 = 0; -/// The M3 syscall surface, dispatched on the saved syscall number: -/// 0 = exit(code) — end the caller (process: free its AS + reschedule; -/// borrowed test thread: unwind to the kernel caller) -/// 1 = ping(value) — record the value + the caller's privilege + the tick count -/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT: -/// 3 = sleep(ms) — block the caller for ms milliseconds -/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md) -/// replaces these later; the result is written back into the trap frame already, -/// since the entry paths restore user registers from it. One handler serves both -/// syscall entry paths. +/// The syscall surface, dispatched on the saved syscall number (`danos.Syscall`). +/// This is the microkernel-minimal set — memory + scheduling only; file/device +/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is +/// written back into the trap frame, since the entry paths restore user registers +/// from it. One handler serves both the syscall/sysret and int-0x80 entry paths. /// /// Install it once at boot (before any user code runs) via `init`. pub fn init() void { arch.setSyscallHandler(syscall); } +/// Return -1 (as an unsigned bit pattern) in the syscall result register. +fn fail(state: *arch.CpuState) void { + arch.setSyscallResult(state, @bitCast(@as(i64, -1))); +} + fn syscall(state: *arch.CpuState) void { - switch (arch.syscallNumber(state)) { - 0 => { + switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) { + .exit => { exit_code = arch.syscallArg(state, 0); // A scheduled process frees its address space and reschedules; a // borrowed test thread unwinds back to the kernel that entered it. if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit(); }, - 3 => { + .yield => { + sched.yield(); + arch.setSyscallResult(state, 0); + }, + .sleep => { sched.sleep(arch.syscallArg(state, 0)); arch.setSyscallResult(state, 0); }, - 1 => { - if (ping_count < pings.len) { - pings[ping_count] = .{ .value = arch.syscallArg(state, 0), .from_user = arch.fromUser(state), .ticks = arch.ticks() }; - ping_count += 1; - } - arch.setSyscallResult(state, 0); - }, - 2 => { - // The pointer must lie inside the user region (image + stack) — - // kernel addresses and non-canonical values fall outside it, so the - // kernel-side read below can't be steered at kernel data. Length is - // checked first so the upper-bound subtraction can't underflow. - // Known gap (fine for a trusted init): a pointer into an *unmapped* - // hole in the region passes the check and the read #PFs -> on_fault - // halts — a self-DoS, not an isolation break. Copy-in with fault - // recovery is M3+ (with SMAP, once there's a reason to enable it). - const ptr = arch.syscallArg(state, 0); - const len = arch.syscallArg(state, 1); - if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) { - const src: [*]const u8 = @ptrFromInt(ptr); - @memcpy(write_buf[0..len], src[0..len]); // keep the latest message - write_len = len; - write_from_user = arch.fromUser(state); - write_count += 1; - log.write("DANOS-INIT: "); - log.write(src[0..len]); - arch.setSyscallResult(state, len); - } else { - arch.setSyscallResult(state, @bitCast(@as(i64, -1))); - } - }, - else => arch.setSyscallResult(state, @bitCast(@as(i64, -1))), + .debug_write => sysDebugWrite(state), + .mmap => sysMmap(state), + .munmap => sysMunmap(state), + _ => fail(state), } } +/// debug_write(ptr, len): copy bytes from user memory into the kernel log. +/// A bring-up diagnostic — real output goes through the VFS/console later. +/// +/// The pointer must lie in the user (low) half, so kernel addresses and +/// non-canonical values fall outside it and the read below can't be steered at +/// kernel data. Length is checked first so the upper-bound add can't overflow. +/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in +/// the user half passes the check and the read #PFs -> on_fault halts — a +/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item. +fn sysDebugWrite(state: *arch.CpuState) void { + const ptr = arch.syscallArg(state, 0); + const len = arch.syscallArg(state, 1); + if (len <= write_buf.len and ptr < user_half_end and ptr + len <= user_half_end) { + const src: [*]const u8 = @ptrFromInt(ptr); + @memcpy(write_buf[0..len], src[0..len]); // keep the latest message + write_len = len; + write_from_user = arch.fromUser(state); + write_count += 1; + log.write("DANOS-INIT: "); + log.write(src[0..len]); + arch.setSyscallResult(state, len); + } else { + fail(state); + } +} + +/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of +/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the +/// base virtual address. `prot` is accepted but not yet honoured (grants are +/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns +/// -1. The user-space allocator (lib `rt`) carves these pages into malloc blocks. +fn sysMmap(state: *arch.CpuState) void { + const len = arch.syscallArg(state, 0); + const t = sched.cur(); + if (t.aspace == 0) return fail(state); // not a user process — nothing to map into + const pages = (len + page_size - 1) / page_size; + if (pages == 0 or pages > max_mmap_pages) return fail(state); + + if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily + const base = t.heap_next; + if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted + + // Reserve all frames up front so a mid-way exhaustion rolls back cleanly + // (no partially-mapped grant leaks into the address space). + var frames: [max_mmap_pages]u64 = undefined; + var got: usize = 0; + while (got < pages) : (got += 1) { + frames[got] = pmm.alloc() orelse { + for (frames[0..got]) |f| pmm.free(f); + return fail(state); + }; + } + + for (frames[0..pages], 0..) |frame, i| { + const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame)); + @memset(dst[0..page_size], 0); // hand out zeroed memory + arch.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX + } + t.heap_next = base + pages * page_size; + arch.setSyscallResult(state, base); +} + +/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps +/// each page and frees its frame. The arena is a bump allocator, so the virtual +/// range is not recycled (the user-space allocator reuses freed *blocks* itself); +/// this just returns the physical frames to the kernel. Returns 0, or -1 if the +/// range is not page-aligned or lies outside the arena. +fn sysMunmap(state: *arch.CpuState) void { + const base = arch.syscallArg(state, 0); + const len = arch.syscallArg(state, 1); + const t = sched.cur(); + if (t.aspace == 0 or base % page_size != 0) return fail(state); + const pages = (len + page_size - 1) / page_size; + if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state); + + for (0..pages) |i| { + const va = base + i * page_size; + if (arch.translate(t.aspace, va)) |phys| { + arch.unmapUserPageInto(t.aspace, va); + pmm.free(phys); + } + } + arch.setSyscallResult(state, 0); +} + /// Reset the recorded syscall evidence before a user-mode run. fn resetRecords() void { - ping_count = 0; write_len = 0; write_from_user = false; write_count = 0; diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 8a6eaf6..15fc47d 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -32,7 +32,7 @@ const max_tasks = config.max_tasks; // maximum tasks alive at once (static pool) const State = enum { free, ready, running, blocked }; -const Task = struct { +pub const Task = struct { id: u32 = 0, state: State = .free, priority: Priority = 0, @@ -46,6 +46,10 @@ const Task = struct { aspace: u64 = 0, user_ip: u64 = 0, // user-mode entry point (user task only) user_sp: u64 = 0, // user-mode stack pointer (user task only) + // Next free virtual address in this task's mmap grant arena (0 = uninitialised; + // process.zig lazily seeds it to the arena base on the first mmap). Bumped up + // as the user heap grows; user task only. + heap_next: u64 = 0, next: ?*Task = null, // ready-queue link }; @@ -87,7 +91,7 @@ inline fn thisCpu() *PerCpu { /// The task running on this core — the per-CPU replacement for the old global /// `current`. A convenience reader; writes go through `thisCpu().current`. -inline fn cur() *Task { +pub inline fn cur() *Task { return thisCpu().current; } diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index fea9d4f..e5836f6 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -17,7 +17,7 @@ const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const sched = @import("scheduler.zig"); const ipc = @import("ipc.zig"); -const usermode = @import("usermode.zig"); +const process = @import("process.zig"); /// Formatted write straight to serial, independent of the framebuffer console. fn log(comptime fmt: []const u8, args: anytype) void { @@ -93,8 +93,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { faultNoExecute(); } else if (eql(case, "fault-null")) { faultNull(); - } else if (eql(case, "user")) { - userTest(); + } else if (eql(case, "usermem")) { + userMemTest(); } else if (eql(case, "user-pf")) { userPfTest(); } else if (eql(case, "init")) { @@ -713,24 +713,64 @@ fn smpRetryTest() void { // --- ring 3 (user mode) ----------------------------------------------------- -/// The full ring-3 round trip: enter user mode, take syscalls and timer -/// interrupts from CPL 3, and come back. Preemption is disabled for the run — -/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate -/// (interrupts still fire and iretq back into ring 3, which is the point). -fn userTest() void { - log("DANOS-TEST-BEGIN: user\n", .{}); - sched.setPreemption(false); - const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: { - log("DANOS-USER: run failed: {s}\n", .{@errorName(err)}); - break :blk false; - }; - sched.setPreemption(true); +/// The mmap/munmap grant path: create a fresh address space, hand out pages into +/// its mmap arena the way the `mmap` syscall does, prove they are real (write and +/// read them back through the physmap), then release them via the `munmap` path +/// (translate -> unmap -> free) and tear the address space down. The frame count +/// must return exactly to where it started — a leak or a double-free would show +/// as drift. Exercises the new `translate`/`unmapUserPageInto` primitives that +/// back munmap, without needing a user binary that calls the syscalls. +fn userMemTest() void { + log("DANOS-TEST-BEGIN: usermem\n", .{}); + const base_free = pmm.stats().free_frames; - check("user program ran and exited (ring-3 round trip)", ran); - check("two ping syscalls received", usermode.ping_count == 2); - check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF); - check("syscalls came from user mode (CPL 3)", usermode.pings[0].from_user and usermode.pings[1].from_user); - check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks); + const aspace = arch.createAddressSpace() orelse { + check("created a fresh address space", false); + result(); + return; + }; + check("created a fresh address space", aspace != 0); + + // Grant three pages into the arena, mapped RW + NX (the mmap contract). + const npages = 3; + const arena = process.heap_arena_base; + var frames: [npages]u64 = undefined; + var mapped: usize = 0; + while (mapped < npages) : (mapped += 1) { + frames[mapped] = pmm.alloc() orelse break; + arch.mapUserPageInto(aspace, arena + mapped * danos.page_size, frames[mapped], true, false); + } + check("granted three user pages", mapped == npages); + + // Each page resolves back to the frame we mapped, and is writable RAM. + var translate_ok = true; + var rw_ok = true; + for (0..npages) |i| { + const va = arena + i * danos.page_size; + const phys = arch.translate(aspace, va) orelse { + translate_ok = false; + continue; + }; + if (phys != frames[i]) translate_ok = false; + const p: [*]u8 = @ptrFromInt(danos.physToVirt(phys)); + p[0] = 0xA5; + if (p[0] != 0xA5) rw_ok = false; + } + check("translate resolves each grant to its frame", translate_ok); + check("granted pages are writable RAM", rw_ok); + + // Release them the way munmap does, then tear down the address space. + for (0..npages) |i| { + const va = arena + i * danos.page_size; + if (arch.translate(aspace, va)) |phys| { + arch.unmapUserPageInto(aspace, va); + pmm.free(phys); + } + } + check("munmap unmapped every grant", arch.translate(aspace, arena) == null); + arch.destroyAddressSpace(aspace); + + check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free); result(); } @@ -761,28 +801,28 @@ fn processTest(boot_info: *const BootInfo) void { } const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; - usermode.write_count = 0; - usermode.write_from_user = false; + process.write_count = 0; + process.write_from_user = false; proc_worker_run = true; proc_worker_ran = false; sched.spawn(procWorker, 4); // kernel task at the processes' priority var spawned: u32 = 0; - if (usermode.spawnProcess(image, 4)) spawned += 1 else |_| {} - if (usermode.spawnProcess(image, 4)) spawned += 1 else |_| {} + if (process.spawnProcess(image, 4)) spawned += 1 else |_| {} + if (process.spawnProcess(image, 4)) spawned += 1 else |_| {} // Wait (real time) for several heartbeats across the two processes. Each // process sleeps ~1 s between beats, so a few seconds yields several. sched.setPriority(1); // drop below the workers so they get the cores const deadline = arch.millis() + 8000; - while (usermode.write_count < 4 and arch.millis() < deadline) sched.yield(); + while (process.write_count < 4 and arch.millis() < deadline) sched.yield(); sched.setPriority(4); proc_worker_run = false; - log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, usermode.write_count }); + log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, process.write_count }); check("both init processes spawned on their own address spaces", spawned == 2); - check("processes made repeated heartbeat syscalls (>=4)", usermode.write_count >= 4); - check("heartbeats came from user mode (CPL 3)", usermode.write_from_user); + check("processes made repeated heartbeat syscalls (>=4)", process.write_count >= 4); + check("heartbeats came from user mode (CPL 3)", process.write_from_user); check("a kernel task coexisted with the processes (preemption)", proc_worker_ran); result(); } @@ -794,7 +834,7 @@ fn processTest(boot_info: *const BootInfo) void { fn userPfTest() void { log("DANOS-TEST-BEGIN: user-pf\n", .{}); sched.setPreemption(false); - _ = usermode.run(usermode.pfBlob()) catch {}; + _ = process.run(process.pfBlob()) catch {}; log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{}); } @@ -811,8 +851,8 @@ fn initTest(boot_info: *const BootInfo) void { return; } const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; - usermode.write_count = 0; - const spawned = if (usermode.spawnProcess(image, 4)) true else |err| blk: { + process.write_count = 0; + const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: { log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)}); break :blk false; }; @@ -822,15 +862,15 @@ fn initTest(boot_info: *const BootInfo) void { // and sleeps repeatedly (init sleeps ~1 s between beats). sched.setPriority(1); const deadline = arch.millis() + 8000; - while (usermode.write_count < 2 and arch.millis() < deadline) sched.yield(); + while (process.write_count < 2 and arch.millis() < deadline) sched.yield(); sched.setPriority(4); const prefix = "init: heartbeat"; - const beat_ok = usermode.write_len >= prefix.len and eql(usermode.write_buf[0..prefix.len], prefix); - check("init produced repeated heartbeats (>=2)", usermode.write_count >= 2); + const beat_ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix); + check("init produced repeated heartbeats (>=2)", process.write_count >= 2); check("heartbeat text arrived intact", beat_ok); - check("heartbeats came from user mode (CPL 3)", usermode.write_from_user); - check("init is still alive (did not exit)", usermode.exit_code == 0); + check("heartbeats came from user mode (CPL 3)", process.write_from_user); + check("init is still alive (did not exit)", process.exit_code == 0); result(); } diff --git a/src/root.zig b/src/root.zig index 07cd929..b328ba4 100644 --- a/src/root.zig +++ b/src/root.zig @@ -58,6 +58,26 @@ pub const page_size = 4096; pub const physmap_base: u64 = 0xFFFF_8800_0000_0000; pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000; +/// The kernel syscall numbers — the single source of truth shared by the kernel +/// dispatcher (src/kernel/process.zig) and the user runtime library, so the two +/// can never drift. The set is deliberately microkernel-minimal: file/device I/O +/// is not here — it lives in user-space servers reached through the IPC calls. +/// The table grows one milestone at a time; see docs/syscall.md. +pub const Syscall = enum(u64) { + exit = 0, // exit(code): end the calling process + yield = 1, // yield(): give up the rest of this quantum + debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only) + sleep = 3, // sleep(ms): block the caller for ms milliseconds + mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages + munmap = 5, // munmap(base, len): release pages from a prior mmap + _, +}; + +/// Protection flags for `mmap` (matching the usual C bit values). +pub const prot_read: u64 = 1; +pub const prot_write: u64 = 2; +pub const prot_exec: u64 = 4; + /// Physical address -> its virtual address in the physmap. The single way the /// kernel dereferences a physical address once paging is up. /// diff --git a/test/qemu_test.py b/test/qemu_test.py index bb7228b..5eec46f 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -150,9 +150,9 @@ CASES = [ "expect": r"page fault \(vector 14\)", "fail": r"NX not enforced"}, {"name": "fault-null", "expect": r"page fault \(vector 14\)"}, - # Ring 3: a user program runs at CPL 3, makes int 0x80 syscalls, survives - # timer interrupts, and exits back into the kernel. - {"name": "user", + # Memory grants: the mmap/munmap path hands out user pages into a process's + # arena, translate resolves them, munmap frees them, and no frames leak. + {"name": "usermem", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, # Isolation: a ring-3 read of a kernel-only page must #PF with error code