M5: rename usermode->process, add yield + mmap/munmap syscalls

Start the user-space driver track (VFS + IPC + heap). This lays the
process/syscall foundation the runtime heap will grow on.

- Rename usermode.zig -> process.zig; drop the retired hello/ping blob and
  its `user` test (subsumed by the real /sbin/init exerciser). Keep the
  isolation-proof pf blob and the user-pf test.
- Add danos.Syscall as the single source of truth for syscall numbers, shared
  by the kernel dispatcher and (later) the user runtime lib. Dispatch on the
  enum. New calls: 1=yield, 4=mmap, 5=munmap. Widen debug_write's bounds
  check to the whole user low half so heap buffers are writable.
- mmap grants zeroed RW+NX pages from a per-process bump arena
  (Task.heap_next, PML4[224] above image+stack); munmap frees the frames.
  Add paging.translateIn / unmapInto (+ arch.translate / unmapUserPageInto)
  as the primitives munmap and future cross-AS copies need.
- New `usermem` test: grant three pages into a fresh AS, translate them,
  release via the munmap path, tear down, and assert no frames leak.
  Suite 28/28 (user -> usermem).
This commit is contained in:
Daniel Samson
2026-07-09 06:55:09 +01:00
parent 7384a730be
commit 9316f9f1c3
12 changed files with 290 additions and 139 deletions
+1 -1
View File
@@ -152,7 +152,7 @@ pub fn build(b: *std.Build) void {
// --- /sbin/init: the first user-space program ---
// Its own tiny freestanding binary, linked at a fixed address inside the
// kernel's user region (usermode.zig) and started in ring 3 by the kernel's
// kernel's user region (process.zig) and started in ring 3 by the kernel's
// user-ELF loader. `.large` because the image base is above 4 GiB — small/
// medium code models emit 32-bit absolute relocations that can't reach.
// Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its
+1 -1
View File
@@ -6,7 +6,7 @@ System calls (syscalls) are the bridge between your programs and the operating s
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/usermode.zig`); the
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/process.zig`); the
> handler dispatches on whether the caller is a scheduled process (its own address
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
+2 -2
View File
@@ -1,7 +1,7 @@
//! /sbin/init — the first user-space program, PID 1. Built as its own
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
//! kernel (src/kernel/usermode.zig). It talks to the kernel only through the
//! kernel (src/kernel/process.zig). It talks to the kernel only through the
//! `syscall` instruction.
//!
//! Today it's a heartbeat: it prints a line and sleeps, forever — enough to show
@@ -11,7 +11,7 @@
const std = @import("std");
// Syscall numbers (see src/kernel/usermode.zig):
// Syscall numbers (see src/kernel/process.zig):
const sys_exit = 0;
const sys_write = 2;
const sys_sleep = 3;
+13
View File
@@ -190,6 +190,19 @@ pub fn unmapPage(virt: u64) void {
paging.unmap(virt);
}
/// Remove a page mapping from address space `root` (for munmap of user pages).
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
pub fn unmapUserPageInto(root: u64, virt: u64) void {
paging.unmapInto(root, virt);
}
/// Resolve `virt` to its physical address in the address space rooted at `root`
/// (any address space, not just the live one), or null if unmapped. Used to find
/// the frame behind a user page for munmap, and for cross-address-space copies.
pub fn translate(root: u64, virt: u64) ?u64 {
return paging.translateIn(root, virt);
}
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
/// W^X: code read-only + executable, data writable + no-execute.
pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void {
+4 -24
View File
@@ -235,34 +235,14 @@ user_saved_rsp:
.skip 8
.text
# --- user-mode test programs -------------------------------------------------
# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and
# --- user-mode test program --------------------------------------------------
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
# entered via enter_user. Position-independent (immediates and short jumps only).
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
# from the user mapping.
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
# became the real ring-3 exerciser; only the isolation proof remains.)
.section .rodata
# The hello program: ping syscall (rax=1) with a value, a delay loop long enough
# for several timer ticks to land while at CPL 3 (proving interrupt-from-user +
# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety
# net in case exit ever returns.
.global user_prog_start
.global user_prog_end
user_prog_start:
mov $1, %rax
mov $0xC0DE, %rdi
int $0x80
mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG
1: dec %rcx
jnz 1b
mov $1, %rax
mov $0xBEEF, %rdi
int $0x80
mov $0, %rax
int $0x80
2: jmp 2b
user_prog_end:
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
# page, so this must take a #PF with error code 0x5 (present | user) before any
+27 -1
View File
@@ -306,7 +306,15 @@ pub fn setExecutable(phys: u64) void {
/// Remove a mapping and flush it from the TLB.
pub fn unmap(virt: u64) void {
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
unmapInto(kernel_pml4, virt);
}
/// Remove a mapping from the address space rooted at `pml4` (a process's own
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
/// the intermediate tables and any frame the PTE pointed at are left to the
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
pub fn unmapInto(pml4: u64, virt: u64) void {
const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF];
if (pml4e & present == 0) return;
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
if (pdpte & present == 0) return;
@@ -316,6 +324,24 @@ pub fn unmap(virt: u64) void {
invalidate(virt);
}
/// Resolve a virtual address to a physical one in the address space rooted at
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
/// any address space, not just the live one). Returns null if `virt` is not
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
/// case. The foundation for cross-address-space copies and for munmap (which
/// needs the frame behind a user vaddr to free it).
pub fn translateIn(pml4: u64, virt: u64) ?u64 {
const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF];
if (pml4e & present == 0) return null;
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
if (pdpte & present == 0) return null;
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
if (pde & present == 0) return null;
const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF];
if (pte & present == 0) return null;
return (pte & addr_mask) | (virt & (page_size - 1));
}
fn invalidate(virt: u64) void {
// invlpg needs its operand via a register-indirect memory reference that Zig
// inline asm won't form directly, so stage the address in a register first.
+3 -3
View File
@@ -7,7 +7,7 @@ const log = @import("log.zig");
const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const scheduler = @import("scheduler.zig");
const usermode = @import("usermode.zig");
const process = @import("process.zig");
const platform = @import("platform");
const tests = @import("tests.zig");
const build_options = @import("build_options");
@@ -228,7 +228,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
// Install the syscall handler (int 0x80 gate + syscall stub) once, before any
// user code runs.
usermode.init();
process.init();
// Register the current context as the first task before enabling preemption.
scheduler.init(4);
@@ -263,7 +263,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
if (boot_info.init_len != 0) {
status("starting /sbin/init...\n");
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
usermode.spawnProcess(image, 4) catch |err| {
process.spawnProcess(image, 4) catch |err| {
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
};
} else {
@@ -1,10 +1,14 @@
//! Ring-3 execution. Two entry points:
//! - `spawnProcess` loads a user ELF (`/sbin/init`) into a fresh address space
//! and schedules it as a real preemptive ring-3 process on its own page
//! tables. This is the production path.
//! - `run` executes a raw code blob (the user/user-pf test programs) on the
//! User-space processes: loading a user ELF and running it in ring 3. danos is a
//! microkernel, so this only ever loads *user* binaries — there is no kernel-space
//! loader; in-kernel code is linked into the kernel image, not loaded here.
//!
//! Two entry points:
//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers)
//! into a fresh address space and schedules it as a real preemptive ring-3
//! process on its own page tables. This is the production path.
//! - `run` executes a raw code blob (the user-pf isolation test program) on the
//! *current* kernel context via the borrowed-thread path — a minimal probe of
//! the ring-transition mechanisms, kept for the tests.
//! the ring-transition mechanisms, kept for that test.
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
//! talks to the kernel only through the syscall instruction (or the int 0x80
//! gate). The shared handler is installed once by `init`.
@@ -25,6 +29,7 @@ const sync = @import("sync.zig");
const log = @import("log.zig");
const page_size = danos.page_size;
const Syscall = danos.Syscall;
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
/// the identity map (low indices) and the vmm test address (index 128), so
@@ -33,100 +38,163 @@ const page_size = danos.page_size;
pub const code_virt: u64 = 0x0000_7000_0000_0000;
pub const stack_virt: u64 = 0x0000_7000_0020_0000;
// The hand-assembled user program blobs (isr.s, .rodata).
const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" });
const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" });
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
/// 1 GiB window is far more than any user heap needs today.
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
/// used to bound the addresses a syscall will dereference on the caller's behalf.
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
/// chunks, so this bound is generous; it also caps the frame scratch array below.
const max_mmap_pages = 256;
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit.
pub fn helloBlob() []const u8 {
return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)];
}
/// The isolation-proof program: reads a kernel-only page, must #PF.
pub fn pfBlob() []const u8 {
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
}
/// What a ping syscall recorded — evidence for the test to assert on.
pub const Ping = struct { value: u64 = 0, from_user: bool = false, ticks: u64 = 0 };
pub var pings = [_]Ping{.{}} ** 2;
pub var ping_count: usize = 0;
/// What write syscalls produced (accumulated), and the exit syscall's code.
/// What debug_write syscalls produced (accumulated), and the exit syscall's code.
pub var write_buf: [256]u8 = undefined;
pub var write_len: usize = 0;
pub var write_from_user: bool = false;
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
pub var exit_code: u64 = 0;
/// The M3 syscall surface, dispatched on the saved syscall number:
/// 0 = exit(code) — end the caller (process: free its AS + reschedule;
/// borrowed test thread: unwind to the kernel caller)
/// 1 = ping(value) — record the value + the caller's privilege + the tick count
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
/// 3 = sleep(ms) — block the caller for ms milliseconds
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
/// replaces these later; the result is written back into the trap frame already,
/// since the entry paths restore user registers from it. One handler serves both
/// syscall entry paths.
/// The syscall surface, dispatched on the saved syscall number (`danos.Syscall`).
/// This is the microkernel-minimal set — memory + scheduling only; file/device
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
/// written back into the trap frame, since the entry paths restore user registers
/// from it. One handler serves both the syscall/sysret and int-0x80 entry paths.
///
/// Install it once at boot (before any user code runs) via `init`.
pub fn init() void {
arch.setSyscallHandler(syscall);
}
/// Return -1 (as an unsigned bit pattern) in the syscall result register.
fn fail(state: *arch.CpuState) void {
arch.setSyscallResult(state, @bitCast(@as(i64, -1)));
}
fn syscall(state: *arch.CpuState) void {
switch (arch.syscallNumber(state)) {
0 => {
switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) {
.exit => {
exit_code = arch.syscallArg(state, 0);
// A scheduled process frees its address space and reschedules; a
// borrowed test thread unwinds back to the kernel that entered it.
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
},
3 => {
.yield => {
sched.yield();
arch.setSyscallResult(state, 0);
},
.sleep => {
sched.sleep(arch.syscallArg(state, 0));
arch.setSyscallResult(state, 0);
},
1 => {
if (ping_count < pings.len) {
pings[ping_count] = .{ .value = arch.syscallArg(state, 0), .from_user = arch.fromUser(state), .ticks = arch.ticks() };
ping_count += 1;
}
arch.setSyscallResult(state, 0);
},
2 => {
// The pointer must lie inside the user region (image + stack) —
// kernel addresses and non-canonical values fall outside it, so the
// kernel-side read below can't be steered at kernel data. Length is
// checked first so the upper-bound subtraction can't underflow.
// Known gap (fine for a trusted init): a pointer into an *unmapped*
// hole in the region passes the check and the read #PFs -> on_fault
// halts — a self-DoS, not an isolation break. Copy-in with fault
// recovery is M3+ (with SMAP, once there's a reason to enable it).
const ptr = arch.syscallArg(state, 0);
const len = arch.syscallArg(state, 1);
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
const src: [*]const u8 = @ptrFromInt(ptr);
@memcpy(write_buf[0..len], src[0..len]); // keep the latest message
write_len = len;
write_from_user = arch.fromUser(state);
write_count += 1;
log.write("DANOS-INIT: ");
log.write(src[0..len]);
arch.setSyscallResult(state, len);
} else {
arch.setSyscallResult(state, @bitCast(@as(i64, -1)));
}
},
else => arch.setSyscallResult(state, @bitCast(@as(i64, -1))),
.debug_write => sysDebugWrite(state),
.mmap => sysMmap(state),
.munmap => sysMunmap(state),
_ => fail(state),
}
}
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
/// A bring-up diagnostic — real output goes through the VFS/console later.
///
/// The pointer must lie in the user (low) half, so kernel addresses and
/// non-canonical values fall outside it and the read below can't be steered at
/// kernel data. Length is checked first so the upper-bound add can't overflow.
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
/// the user half passes the check and the read #PFs -> on_fault halts — a
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
fn sysDebugWrite(state: *arch.CpuState) void {
const ptr = arch.syscallArg(state, 0);
const len = arch.syscallArg(state, 1);
if (len <= write_buf.len and ptr < user_half_end and ptr + len <= user_half_end) {
const src: [*]const u8 = @ptrFromInt(ptr);
@memcpy(write_buf[0..len], src[0..len]); // keep the latest message
write_len = len;
write_from_user = arch.fromUser(state);
write_count += 1;
log.write("DANOS-INIT: ");
log.write(src[0..len]);
arch.setSyscallResult(state, len);
} else {
fail(state);
}
}
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
/// base virtual address. `prot` is accepted but not yet honoured (grants are
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
/// -1. The user-space allocator (lib `rt`) carves these pages into malloc blocks.
fn sysMmap(state: *arch.CpuState) void {
const len = arch.syscallArg(state, 0);
const t = sched.cur();
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
const pages = (len + page_size - 1) / page_size;
if (pages == 0 or pages > max_mmap_pages) return fail(state);
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
const base = t.heap_next;
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
// (no partially-mapped grant leaks into the address space).
var frames: [max_mmap_pages]u64 = undefined;
var got: usize = 0;
while (got < pages) : (got += 1) {
frames[got] = pmm.alloc() orelse {
for (frames[0..got]) |f| pmm.free(f);
return fail(state);
};
}
for (frames[0..pages], 0..) |frame, i| {
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
@memset(dst[0..page_size], 0); // hand out zeroed memory
arch.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
}
t.heap_next = base + pages * page_size;
arch.setSyscallResult(state, base);
}
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
/// each page and frees its frame. The arena is a bump allocator, so the virtual
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
/// range is not page-aligned or lies outside the arena.
fn sysMunmap(state: *arch.CpuState) void {
const base = arch.syscallArg(state, 0);
const len = arch.syscallArg(state, 1);
const t = sched.cur();
if (t.aspace == 0 or base % page_size != 0) return fail(state);
const pages = (len + page_size - 1) / page_size;
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
for (0..pages) |i| {
const va = base + i * page_size;
if (arch.translate(t.aspace, va)) |phys| {
arch.unmapUserPageInto(t.aspace, va);
pmm.free(phys);
}
}
arch.setSyscallResult(state, 0);
}
/// Reset the recorded syscall evidence before a user-mode run.
fn resetRecords() void {
ping_count = 0;
write_len = 0;
write_from_user = false;
write_count = 0;
+6 -2
View File
@@ -32,7 +32,7 @@ const max_tasks = config.max_tasks; // maximum tasks alive at once (static pool)
const State = enum { free, ready, running, blocked };
const Task = struct {
pub const Task = struct {
id: u32 = 0,
state: State = .free,
priority: Priority = 0,
@@ -46,6 +46,10 @@ const Task = struct {
aspace: u64 = 0,
user_ip: u64 = 0, // user-mode entry point (user task only)
user_sp: u64 = 0, // user-mode stack pointer (user task only)
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
// as the user heap grows; user task only.
heap_next: u64 = 0,
next: ?*Task = null, // ready-queue link
};
@@ -87,7 +91,7 @@ inline fn thisCpu() *PerCpu {
/// The task running on this core — the per-CPU replacement for the old global
/// `current`. A convenience reader; writes go through `thisCpu().current`.
inline fn cur() *Task {
pub inline fn cur() *Task {
return thisCpu().current;
}
+76 -36
View File
@@ -17,7 +17,7 @@ const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const sched = @import("scheduler.zig");
const ipc = @import("ipc.zig");
const usermode = @import("usermode.zig");
const process = @import("process.zig");
/// Formatted write straight to serial, independent of the framebuffer console.
fn log(comptime fmt: []const u8, args: anytype) void {
@@ -93,8 +93,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
faultNoExecute();
} else if (eql(case, "fault-null")) {
faultNull();
} else if (eql(case, "user")) {
userTest();
} else if (eql(case, "usermem")) {
userMemTest();
} else if (eql(case, "user-pf")) {
userPfTest();
} else if (eql(case, "init")) {
@@ -713,24 +713,64 @@ fn smpRetryTest() void {
// --- ring 3 (user mode) -----------------------------------------------------
/// The full ring-3 round trip: enter user mode, take syscalls and timer
/// interrupts from CPL 3, and come back. Preemption is disabled for the run —
/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate
/// (interrupts still fire and iretq back into ring 3, which is the point).
fn userTest() void {
log("DANOS-TEST-BEGIN: user\n", .{});
sched.setPreemption(false);
const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: {
log("DANOS-USER: run failed: {s}\n", .{@errorName(err)});
break :blk false;
};
sched.setPreemption(true);
/// The mmap/munmap grant path: create a fresh address space, hand out pages into
/// its mmap arena the way the `mmap` syscall does, prove they are real (write and
/// read them back through the physmap), then release them via the `munmap` path
/// (translate -> unmap -> free) and tear the address space down. The frame count
/// must return exactly to where it started — a leak or a double-free would show
/// as drift. Exercises the new `translate`/`unmapUserPageInto` primitives that
/// back munmap, without needing a user binary that calls the syscalls.
fn userMemTest() void {
log("DANOS-TEST-BEGIN: usermem\n", .{});
const base_free = pmm.stats().free_frames;
check("user program ran and exited (ring-3 round trip)", ran);
check("two ping syscalls received", usermode.ping_count == 2);
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
check("syscalls came from user mode (CPL 3)", usermode.pings[0].from_user and usermode.pings[1].from_user);
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
const aspace = arch.createAddressSpace() orelse {
check("created a fresh address space", false);
result();
return;
};
check("created a fresh address space", aspace != 0);
// Grant three pages into the arena, mapped RW + NX (the mmap contract).
const npages = 3;
const arena = process.heap_arena_base;
var frames: [npages]u64 = undefined;
var mapped: usize = 0;
while (mapped < npages) : (mapped += 1) {
frames[mapped] = pmm.alloc() orelse break;
arch.mapUserPageInto(aspace, arena + mapped * danos.page_size, frames[mapped], true, false);
}
check("granted three user pages", mapped == npages);
// Each page resolves back to the frame we mapped, and is writable RAM.
var translate_ok = true;
var rw_ok = true;
for (0..npages) |i| {
const va = arena + i * danos.page_size;
const phys = arch.translate(aspace, va) orelse {
translate_ok = false;
continue;
};
if (phys != frames[i]) translate_ok = false;
const p: [*]u8 = @ptrFromInt(danos.physToVirt(phys));
p[0] = 0xA5;
if (p[0] != 0xA5) rw_ok = false;
}
check("translate resolves each grant to its frame", translate_ok);
check("granted pages are writable RAM", rw_ok);
// Release them the way munmap does, then tear down the address space.
for (0..npages) |i| {
const va = arena + i * danos.page_size;
if (arch.translate(aspace, va)) |phys| {
arch.unmapUserPageInto(aspace, va);
pmm.free(phys);
}
}
check("munmap unmapped every grant", arch.translate(aspace, arena) == null);
arch.destroyAddressSpace(aspace);
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
result();
}
@@ -761,28 +801,28 @@ fn processTest(boot_info: *const BootInfo) void {
}
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
usermode.write_count = 0;
usermode.write_from_user = false;
process.write_count = 0;
process.write_from_user = false;
proc_worker_run = true;
proc_worker_ran = false;
sched.spawn(procWorker, 4); // kernel task at the processes' priority
var spawned: u32 = 0;
if (usermode.spawnProcess(image, 4)) spawned += 1 else |_| {}
if (usermode.spawnProcess(image, 4)) spawned += 1 else |_| {}
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
// Wait (real time) for several heartbeats across the two processes. Each
// process sleeps ~1 s between beats, so a few seconds yields several.
sched.setPriority(1); // drop below the workers so they get the cores
const deadline = arch.millis() + 8000;
while (usermode.write_count < 4 and arch.millis() < deadline) sched.yield();
while (process.write_count < 4 and arch.millis() < deadline) sched.yield();
sched.setPriority(4);
proc_worker_run = false;
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, usermode.write_count });
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, process.write_count });
check("both init processes spawned on their own address spaces", spawned == 2);
check("processes made repeated heartbeat syscalls (>=4)", usermode.write_count >= 4);
check("heartbeats came from user mode (CPL 3)", usermode.write_from_user);
check("processes made repeated heartbeat syscalls (>=4)", process.write_count >= 4);
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
check("a kernel task coexisted with the processes (preemption)", proc_worker_ran);
result();
}
@@ -794,7 +834,7 @@ fn processTest(boot_info: *const BootInfo) void {
fn userPfTest() void {
log("DANOS-TEST-BEGIN: user-pf\n", .{});
sched.setPreemption(false);
_ = usermode.run(usermode.pfBlob()) catch {};
_ = process.run(process.pfBlob()) catch {};
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
}
@@ -811,8 +851,8 @@ fn initTest(boot_info: *const BootInfo) void {
return;
}
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
usermode.write_count = 0;
const spawned = if (usermode.spawnProcess(image, 4)) true else |err| blk: {
process.write_count = 0;
const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: {
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
break :blk false;
};
@@ -822,15 +862,15 @@ fn initTest(boot_info: *const BootInfo) void {
// and sleeps repeatedly (init sleeps ~1 s between beats).
sched.setPriority(1);
const deadline = arch.millis() + 8000;
while (usermode.write_count < 2 and arch.millis() < deadline) sched.yield();
while (process.write_count < 2 and arch.millis() < deadline) sched.yield();
sched.setPriority(4);
const prefix = "init: heartbeat";
const beat_ok = usermode.write_len >= prefix.len and eql(usermode.write_buf[0..prefix.len], prefix);
check("init produced repeated heartbeats (>=2)", usermode.write_count >= 2);
const beat_ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix);
check("init produced repeated heartbeats (>=2)", process.write_count >= 2);
check("heartbeat text arrived intact", beat_ok);
check("heartbeats came from user mode (CPL 3)", usermode.write_from_user);
check("init is still alive (did not exit)", usermode.exit_code == 0);
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
check("init is still alive (did not exit)", process.exit_code == 0);
result();
}
+20
View File
@@ -58,6 +58,26 @@ pub const page_size = 4096;
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
/// The kernel syscall numbers — the single source of truth shared by the kernel
/// dispatcher (src/kernel/process.zig) and the user runtime library, so the two
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
/// is not here — it lives in user-space servers reached through the IPC calls.
/// The table grows one milestone at a time; see docs/syscall.md.
pub const Syscall = enum(u64) {
exit = 0, // exit(code): end the calling process
yield = 1, // yield(): give up the rest of this quantum
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
sleep = 3, // sleep(ms): block the caller for ms milliseconds
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
munmap = 5, // munmap(base, len): release pages from a prior mmap
_,
};
/// Protection flags for `mmap` (matching the usual C bit values).
pub const prot_read: u64 = 1;
pub const prot_write: u64 = 2;
pub const prot_exec: u64 = 4;
/// Physical address -> its virtual address in the physmap. The single way the
/// kernel dereferences a physical address once paging is up.
///
+3 -3
View File
@@ -150,9 +150,9 @@ CASES = [
"expect": r"page fault \(vector 14\)",
"fail": r"NX not enforced"},
{"name": "fault-null", "expect": r"page fault \(vector 14\)"},
# Ring 3: a user program runs at CPL 3, makes int 0x80 syscalls, survives
# timer interrupts, and exits back into the kernel.
{"name": "user",
# Memory grants: the mmap/munmap path hands out user pages into a process's
# arena, translate resolves them, munmap frees them, and no frames leak.
{"name": "usermem",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Isolation: a ring-3 read of a kernel-only page must #PF with error code