M5: rename usermode->process, add yield + mmap/munmap syscalls
Start the user-space driver track (VFS + IPC + heap). This lays the process/syscall foundation the runtime heap will grow on. - Rename usermode.zig -> process.zig; drop the retired hello/ping blob and its `user` test (subsumed by the real /sbin/init exerciser). Keep the isolation-proof pf blob and the user-pf test. - Add danos.Syscall as the single source of truth for syscall numbers, shared by the kernel dispatcher and (later) the user runtime lib. Dispatch on the enum. New calls: 1=yield, 4=mmap, 5=munmap. Widen debug_write's bounds check to the whole user low half so heap buffers are writable. - mmap grants zeroed RW+NX pages from a per-process bump arena (Task.heap_next, PML4[224] above image+stack); munmap frees the frames. Add paging.translateIn / unmapInto (+ arch.translate / unmapUserPageInto) as the primitives munmap and future cross-AS copies need. - New `usermem` test: grant three pages into a fresh AS, translate them, release via the munmap path, tear down, and assert no frames leak. Suite 28/28 (user -> usermem).
This commit is contained in:
@@ -152,7 +152,7 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
// --- /sbin/init: the first user-space program ---
|
||||
// Its own tiny freestanding binary, linked at a fixed address inside the
|
||||
// kernel's user region (usermode.zig) and started in ring 3 by the kernel's
|
||||
// kernel's user region (process.zig) and started in ring 3 by the kernel's
|
||||
// user-ELF loader. `.large` because the image base is above 4 GiB — small/
|
||||
// medium code models emit 32-bit absolute relocations that can't reach.
|
||||
// Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ System calls (syscalls) are the bridge between your programs and the operating s
|
||||
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/usermode.zig`); the
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/process.zig`); the
|
||||
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||
|
||||
+2
-2
@@ -1,7 +1,7 @@
|
||||
//! /sbin/init — the first user-space program, PID 1. Built as its own
|
||||
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
|
||||
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
|
||||
//! kernel (src/kernel/usermode.zig). It talks to the kernel only through the
|
||||
//! kernel (src/kernel/process.zig). It talks to the kernel only through the
|
||||
//! `syscall` instruction.
|
||||
//!
|
||||
//! Today it's a heartbeat: it prints a line and sleeps, forever — enough to show
|
||||
@@ -11,7 +11,7 @@
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// Syscall numbers (see src/kernel/usermode.zig):
|
||||
// Syscall numbers (see src/kernel/process.zig):
|
||||
const sys_exit = 0;
|
||||
const sys_write = 2;
|
||||
const sys_sleep = 3;
|
||||
|
||||
@@ -190,6 +190,19 @@ pub fn unmapPage(virt: u64) void {
|
||||
paging.unmap(virt);
|
||||
}
|
||||
|
||||
/// Remove a page mapping from address space `root` (for munmap of user pages).
|
||||
/// Clears the leaf entry only; freeing the underlying frame is the caller's job.
|
||||
pub fn unmapUserPageInto(root: u64, virt: u64) void {
|
||||
paging.unmapInto(root, virt);
|
||||
}
|
||||
|
||||
/// Resolve `virt` to its physical address in the address space rooted at `root`
|
||||
/// (any address space, not just the live one), or null if unmapped. Used to find
|
||||
/// the frame behind a user page for munmap, and for cross-address-space copies.
|
||||
pub fn translate(root: u64, virt: u64) ?u64 {
|
||||
return paging.translateIn(root, virt);
|
||||
}
|
||||
|
||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||
/// W^X: code read-only + executable, data writable + no-execute.
|
||||
pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void {
|
||||
|
||||
@@ -235,34 +235,14 @@ user_saved_rsp:
|
||||
.skip 8
|
||||
.text
|
||||
|
||||
# --- user-mode test programs -------------------------------------------------
|
||||
# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and
|
||||
# --- user-mode test program --------------------------------------------------
|
||||
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
|
||||
# entered via enter_user. Position-independent (immediates and short jumps only).
|
||||
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
|
||||
# from the user mapping.
|
||||
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
|
||||
# became the real ring-3 exerciser; only the isolation proof remains.)
|
||||
.section .rodata
|
||||
|
||||
# The hello program: ping syscall (rax=1) with a value, a delay loop long enough
|
||||
# for several timer ticks to land while at CPL 3 (proving interrupt-from-user +
|
||||
# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety
|
||||
# net in case exit ever returns.
|
||||
.global user_prog_start
|
||||
.global user_prog_end
|
||||
user_prog_start:
|
||||
mov $1, %rax
|
||||
mov $0xC0DE, %rdi
|
||||
int $0x80
|
||||
mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG
|
||||
1: dec %rcx
|
||||
jnz 1b
|
||||
mov $1, %rax
|
||||
mov $0xBEEF, %rdi
|
||||
int $0x80
|
||||
mov $0, %rax
|
||||
int $0x80
|
||||
2: jmp 2b
|
||||
user_prog_end:
|
||||
|
||||
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
||||
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
|
||||
# page, so this must take a #PF with error code 0x5 (present | user) before any
|
||||
|
||||
@@ -306,7 +306,15 @@ pub fn setExecutable(phys: u64) void {
|
||||
|
||||
/// Remove a mapping and flush it from the TLB.
|
||||
pub fn unmap(virt: u64) void {
|
||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
||||
unmapInto(kernel_pml4, virt);
|
||||
}
|
||||
|
||||
/// Remove a mapping from the address space rooted at `pml4` (a process's own
|
||||
/// table or the kernel's) and flush it from the TLB. Clears only the leaf PTE —
|
||||
/// the intermediate tables and any frame the PTE pointed at are left to the
|
||||
/// caller (munmap frees the frame; `destroyAddressSpace` reclaims the tables).
|
||||
pub fn unmapInto(pml4: u64, virt: u64) void {
|
||||
const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return;
|
||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return;
|
||||
@@ -316,6 +324,24 @@ pub fn unmap(virt: u64) void {
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
/// Resolve a virtual address to a physical one in the address space rooted at
|
||||
/// `pml4`, walking the tables through the physmap (CR3-independent — works for
|
||||
/// any address space, not just the live one). Returns null if `virt` is not
|
||||
/// mapped at any level. All danos mappings are 4 KiB, so there is no huge-page
|
||||
/// case. The foundation for cross-address-space copies and for munmap (which
|
||||
/// needs the frame behind a user vaddr to free it).
|
||||
pub fn translateIn(pml4: u64, virt: u64) ?u64 {
|
||||
const pml4e = tableAt(pml4)[(virt >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return null;
|
||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return null;
|
||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return null;
|
||||
const pte = tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF];
|
||||
if (pte & present == 0) return null;
|
||||
return (pte & addr_mask) | (virt & (page_size - 1));
|
||||
}
|
||||
|
||||
fn invalidate(virt: u64) void {
|
||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||
// inline asm won't form directly, so stage the address in a register first.
|
||||
|
||||
+3
-3
@@ -7,7 +7,7 @@ const log = @import("log.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const usermode = @import("usermode.zig");
|
||||
const process = @import("process.zig");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
@@ -228,7 +228,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
|
||||
// Install the syscall handler (int 0x80 gate + syscall stub) once, before any
|
||||
// user code runs.
|
||||
usermode.init();
|
||||
process.init();
|
||||
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
scheduler.init(4);
|
||||
@@ -263,7 +263,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
if (boot_info.init_len != 0) {
|
||||
status("starting /sbin/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||
usermode.spawnProcess(image, 4) catch |err| {
|
||||
process.spawnProcess(image, 4) catch |err| {
|
||||
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
|
||||
@@ -1,10 +1,14 @@
|
||||
//! Ring-3 execution. Two entry points:
|
||||
//! - `spawnProcess` loads a user ELF (`/sbin/init`) into a fresh address space
|
||||
//! and schedules it as a real preemptive ring-3 process on its own page
|
||||
//! tables. This is the production path.
|
||||
//! - `run` executes a raw code blob (the user/user-pf test programs) on the
|
||||
//! User-space processes: loading a user ELF and running it in ring 3. danos is a
|
||||
//! microkernel, so this only ever loads *user* binaries — there is no kernel-space
|
||||
//! loader; in-kernel code is linked into the kernel image, not loaded here.
|
||||
//!
|
||||
//! Two entry points:
|
||||
//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers)
|
||||
//! into a fresh address space and schedules it as a real preemptive ring-3
|
||||
//! process on its own page tables. This is the production path.
|
||||
//! - `run` executes a raw code blob (the user-pf isolation test program) on the
|
||||
//! *current* kernel context via the borrowed-thread path — a minimal probe of
|
||||
//! the ring-transition mechanisms, kept for the tests.
|
||||
//! the ring-transition mechanisms, kept for that test.
|
||||
//! Both map frames user-accessible with W^X (code RO+X, data RW+NX); the program
|
||||
//! talks to the kernel only through the syscall instruction (or the int 0x80
|
||||
//! gate). The shared handler is installed once by `init`.
|
||||
@@ -25,6 +29,7 @@ const sync = @import("sync.zig");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const Syscall = danos.Syscall;
|
||||
|
||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||
@@ -33,81 +38,89 @@ const page_size = danos.page_size;
|
||||
pub const code_virt: u64 = 0x0000_7000_0000_0000;
|
||||
pub const stack_virt: u64 = 0x0000_7000_0020_0000;
|
||||
|
||||
// The hand-assembled user program blobs (isr.s, .rodata).
|
||||
const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" });
|
||||
const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" });
|
||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
|
||||
/// 1 GiB window is far more than any user heap needs today.
|
||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||
|
||||
/// End of the user (low) canonical half. Any legitimate user pointer is below it;
|
||||
/// used to bound the addresses a syscall will dereference on the caller's behalf.
|
||||
pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
|
||||
/// chunks, so this bound is generous; it also caps the frame scratch array below.
|
||||
const max_mmap_pages = 256;
|
||||
|
||||
// The hand-assembled user program blob (isr.s, .rodata) — the isolation probe.
|
||||
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
||||
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
|
||||
|
||||
/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit.
|
||||
pub fn helloBlob() []const u8 {
|
||||
return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)];
|
||||
}
|
||||
|
||||
/// The isolation-proof program: reads a kernel-only page, must #PF.
|
||||
pub fn pfBlob() []const u8 {
|
||||
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
|
||||
}
|
||||
|
||||
/// What a ping syscall recorded — evidence for the test to assert on.
|
||||
pub const Ping = struct { value: u64 = 0, from_user: bool = false, ticks: u64 = 0 };
|
||||
pub var pings = [_]Ping{.{}} ** 2;
|
||||
pub var ping_count: usize = 0;
|
||||
|
||||
/// What write syscalls produced (accumulated), and the exit syscall's code.
|
||||
/// What debug_write syscalls produced (accumulated), and the exit syscall's code.
|
||||
pub var write_buf: [256]u8 = undefined;
|
||||
pub var write_len: usize = 0;
|
||||
pub var write_from_user: bool = false;
|
||||
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
||||
pub var exit_code: u64 = 0;
|
||||
|
||||
/// The M3 syscall surface, dispatched on the saved syscall number:
|
||||
/// 0 = exit(code) — end the caller (process: free its AS + reschedule;
|
||||
/// borrowed test thread: unwind to the kernel caller)
|
||||
/// 1 = ping(value) — record the value + the caller's privilege + the tick count
|
||||
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
|
||||
/// 3 = sleep(ms) — block the caller for ms milliseconds
|
||||
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
|
||||
/// replaces these later; the result is written back into the trap frame already,
|
||||
/// since the entry paths restore user registers from it. One handler serves both
|
||||
/// syscall entry paths.
|
||||
/// The syscall surface, dispatched on the saved syscall number (`danos.Syscall`).
|
||||
/// This is the microkernel-minimal set — memory + scheduling only; file/device
|
||||
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
|
||||
/// written back into the trap frame, since the entry paths restore user registers
|
||||
/// from it. One handler serves both the syscall/sysret and int-0x80 entry paths.
|
||||
///
|
||||
/// Install it once at boot (before any user code runs) via `init`.
|
||||
pub fn init() void {
|
||||
arch.setSyscallHandler(syscall);
|
||||
}
|
||||
|
||||
/// Return -1 (as an unsigned bit pattern) in the syscall result register.
|
||||
fn fail(state: *arch.CpuState) void {
|
||||
arch.setSyscallResult(state, @bitCast(@as(i64, -1)));
|
||||
}
|
||||
|
||||
fn syscall(state: *arch.CpuState) void {
|
||||
switch (arch.syscallNumber(state)) {
|
||||
0 => {
|
||||
switch (@as(Syscall, @enumFromInt(arch.syscallNumber(state)))) {
|
||||
.exit => {
|
||||
exit_code = arch.syscallArg(state, 0);
|
||||
// A scheduled process frees its address space and reschedules; a
|
||||
// borrowed test thread unwinds back to the kernel that entered it.
|
||||
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
|
||||
},
|
||||
3 => {
|
||||
.yield => {
|
||||
sched.yield();
|
||||
arch.setSyscallResult(state, 0);
|
||||
},
|
||||
.sleep => {
|
||||
sched.sleep(arch.syscallArg(state, 0));
|
||||
arch.setSyscallResult(state, 0);
|
||||
},
|
||||
1 => {
|
||||
if (ping_count < pings.len) {
|
||||
pings[ping_count] = .{ .value = arch.syscallArg(state, 0), .from_user = arch.fromUser(state), .ticks = arch.ticks() };
|
||||
ping_count += 1;
|
||||
.debug_write => sysDebugWrite(state),
|
||||
.mmap => sysMmap(state),
|
||||
.munmap => sysMunmap(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
arch.setSyscallResult(state, 0);
|
||||
},
|
||||
2 => {
|
||||
// The pointer must lie inside the user region (image + stack) —
|
||||
// kernel addresses and non-canonical values fall outside it, so the
|
||||
// kernel-side read below can't be steered at kernel data. Length is
|
||||
// checked first so the upper-bound subtraction can't underflow.
|
||||
// Known gap (fine for a trusted init): a pointer into an *unmapped*
|
||||
// hole in the region passes the check and the read #PFs -> on_fault
|
||||
// halts — a self-DoS, not an isolation break. Copy-in with fault
|
||||
// recovery is M3+ (with SMAP, once there's a reason to enable it).
|
||||
}
|
||||
|
||||
/// debug_write(ptr, len): copy bytes from user memory into the kernel log.
|
||||
/// A bring-up diagnostic — real output goes through the VFS/console later.
|
||||
///
|
||||
/// The pointer must lie in the user (low) half, so kernel addresses and
|
||||
/// non-canonical values fall outside it and the read below can't be steered at
|
||||
/// kernel data. Length is checked first so the upper-bound add can't overflow.
|
||||
/// Known gap (fine for trusted user code): a pointer into an *unmapped* hole in
|
||||
/// the user half passes the check and the read #PFs -> on_fault halts — a
|
||||
/// self-DoS, not an isolation break. Fault-recovering copy-in is a later item.
|
||||
fn sysDebugWrite(state: *arch.CpuState) void {
|
||||
const ptr = arch.syscallArg(state, 0);
|
||||
const len = arch.syscallArg(state, 1);
|
||||
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
|
||||
if (len <= write_buf.len and ptr < user_half_end and ptr + len <= user_half_end) {
|
||||
const src: [*]const u8 = @ptrFromInt(ptr);
|
||||
@memcpy(write_buf[0..len], src[0..len]); // keep the latest message
|
||||
write_len = len;
|
||||
@@ -117,16 +130,71 @@ fn syscall(state: *arch.CpuState) void {
|
||||
log.write(src[0..len]);
|
||||
arch.setSyscallResult(state, len);
|
||||
} else {
|
||||
arch.setSyscallResult(state, @bitCast(@as(i64, -1)));
|
||||
fail(state);
|
||||
}
|
||||
},
|
||||
else => arch.setSyscallResult(state, @bitCast(@as(i64, -1))),
|
||||
}
|
||||
|
||||
/// mmap(len, prot) -> base: grant `len` bytes (rounded up to whole pages) of
|
||||
/// fresh, zeroed, writable+NX memory in the caller's mmap arena, and return the
|
||||
/// base virtual address. `prot` is accepted but not yet honoured (grants are
|
||||
/// always RW+NX; W^X for user code stays with the ELF loader). Failure returns
|
||||
/// -1. The user-space allocator (lib `rt`) carves these pages into malloc blocks.
|
||||
fn sysMmap(state: *arch.CpuState) void {
|
||||
const len = arch.syscallArg(state, 0);
|
||||
const t = sched.cur();
|
||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (pages == 0 or pages > max_mmap_pages) return fail(state);
|
||||
|
||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
||||
const base = t.heap_next;
|
||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
|
||||
// Reserve all frames up front so a mid-way exhaustion rolls back cleanly
|
||||
// (no partially-mapped grant leaks into the address space).
|
||||
var frames: [max_mmap_pages]u64 = undefined;
|
||||
var got: usize = 0;
|
||||
while (got < pages) : (got += 1) {
|
||||
frames[got] = pmm.alloc() orelse {
|
||||
for (frames[0..got]) |f| pmm.free(f);
|
||||
return fail(state);
|
||||
};
|
||||
}
|
||||
|
||||
for (frames[0..pages], 0..) |frame, i| {
|
||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
||||
@memset(dst[0..page_size], 0); // hand out zeroed memory
|
||||
arch.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
||||
}
|
||||
t.heap_next = base + pages * page_size;
|
||||
arch.setSyscallResult(state, base);
|
||||
}
|
||||
|
||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||
/// each page and frees its frame. The arena is a bump allocator, so the virtual
|
||||
/// range is not recycled (the user-space allocator reuses freed *blocks* itself);
|
||||
/// this just returns the physical frames to the kernel. Returns 0, or -1 if the
|
||||
/// range is not page-aligned or lies outside the arena.
|
||||
fn sysMunmap(state: *arch.CpuState) void {
|
||||
const base = arch.syscallArg(state, 0);
|
||||
const len = arch.syscallArg(state, 1);
|
||||
const t = sched.cur();
|
||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base + i * page_size;
|
||||
if (arch.translate(t.aspace, va)) |phys| {
|
||||
arch.unmapUserPageInto(t.aspace, va);
|
||||
pmm.free(phys);
|
||||
}
|
||||
}
|
||||
arch.setSyscallResult(state, 0);
|
||||
}
|
||||
|
||||
/// Reset the recorded syscall evidence before a user-mode run.
|
||||
fn resetRecords() void {
|
||||
ping_count = 0;
|
||||
write_len = 0;
|
||||
write_from_user = false;
|
||||
write_count = 0;
|
||||
@@ -32,7 +32,7 @@ const max_tasks = config.max_tasks; // maximum tasks alive at once (static pool)
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
|
||||
const Task = struct {
|
||||
pub const Task = struct {
|
||||
id: u32 = 0,
|
||||
state: State = .free,
|
||||
priority: Priority = 0,
|
||||
@@ -46,6 +46,10 @@ const Task = struct {
|
||||
aspace: u64 = 0,
|
||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
||||
// as the user heap grows; user task only.
|
||||
heap_next: u64 = 0,
|
||||
next: ?*Task = null, // ready-queue link
|
||||
};
|
||||
|
||||
@@ -87,7 +91,7 @@ inline fn thisCpu() *PerCpu {
|
||||
|
||||
/// The task running on this core — the per-CPU replacement for the old global
|
||||
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
||||
inline fn cur() *Task {
|
||||
pub inline fn cur() *Task {
|
||||
return thisCpu().current;
|
||||
}
|
||||
|
||||
|
||||
+76
-36
@@ -17,7 +17,7 @@ const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const usermode = @import("usermode.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
||||
@@ -93,8 +93,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
faultNoExecute();
|
||||
} else if (eql(case, "fault-null")) {
|
||||
faultNull();
|
||||
} else if (eql(case, "user")) {
|
||||
userTest();
|
||||
} else if (eql(case, "usermem")) {
|
||||
userMemTest();
|
||||
} else if (eql(case, "user-pf")) {
|
||||
userPfTest();
|
||||
} else if (eql(case, "init")) {
|
||||
@@ -713,24 +713,64 @@ fn smpRetryTest() void {
|
||||
|
||||
// --- ring 3 (user mode) -----------------------------------------------------
|
||||
|
||||
/// The full ring-3 round trip: enter user mode, take syscalls and timer
|
||||
/// interrupts from CPL 3, and come back. Preemption is disabled for the run —
|
||||
/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate
|
||||
/// (interrupts still fire and iretq back into ring 3, which is the point).
|
||||
fn userTest() void {
|
||||
log("DANOS-TEST-BEGIN: user\n", .{});
|
||||
sched.setPreemption(false);
|
||||
const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: {
|
||||
log("DANOS-USER: run failed: {s}\n", .{@errorName(err)});
|
||||
break :blk false;
|
||||
};
|
||||
sched.setPreemption(true);
|
||||
/// The mmap/munmap grant path: create a fresh address space, hand out pages into
|
||||
/// its mmap arena the way the `mmap` syscall does, prove they are real (write and
|
||||
/// read them back through the physmap), then release them via the `munmap` path
|
||||
/// (translate -> unmap -> free) and tear the address space down. The frame count
|
||||
/// must return exactly to where it started — a leak or a double-free would show
|
||||
/// as drift. Exercises the new `translate`/`unmapUserPageInto` primitives that
|
||||
/// back munmap, without needing a user binary that calls the syscalls.
|
||||
fn userMemTest() void {
|
||||
log("DANOS-TEST-BEGIN: usermem\n", .{});
|
||||
const base_free = pmm.stats().free_frames;
|
||||
|
||||
check("user program ran and exited (ring-3 round trip)", ran);
|
||||
check("two ping syscalls received", usermode.ping_count == 2);
|
||||
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
|
||||
check("syscalls came from user mode (CPL 3)", usermode.pings[0].from_user and usermode.pings[1].from_user);
|
||||
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
|
||||
const aspace = arch.createAddressSpace() orelse {
|
||||
check("created a fresh address space", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
check("created a fresh address space", aspace != 0);
|
||||
|
||||
// Grant three pages into the arena, mapped RW + NX (the mmap contract).
|
||||
const npages = 3;
|
||||
const arena = process.heap_arena_base;
|
||||
var frames: [npages]u64 = undefined;
|
||||
var mapped: usize = 0;
|
||||
while (mapped < npages) : (mapped += 1) {
|
||||
frames[mapped] = pmm.alloc() orelse break;
|
||||
arch.mapUserPageInto(aspace, arena + mapped * danos.page_size, frames[mapped], true, false);
|
||||
}
|
||||
check("granted three user pages", mapped == npages);
|
||||
|
||||
// Each page resolves back to the frame we mapped, and is writable RAM.
|
||||
var translate_ok = true;
|
||||
var rw_ok = true;
|
||||
for (0..npages) |i| {
|
||||
const va = arena + i * danos.page_size;
|
||||
const phys = arch.translate(aspace, va) orelse {
|
||||
translate_ok = false;
|
||||
continue;
|
||||
};
|
||||
if (phys != frames[i]) translate_ok = false;
|
||||
const p: [*]u8 = @ptrFromInt(danos.physToVirt(phys));
|
||||
p[0] = 0xA5;
|
||||
if (p[0] != 0xA5) rw_ok = false;
|
||||
}
|
||||
check("translate resolves each grant to its frame", translate_ok);
|
||||
check("granted pages are writable RAM", rw_ok);
|
||||
|
||||
// Release them the way munmap does, then tear down the address space.
|
||||
for (0..npages) |i| {
|
||||
const va = arena + i * danos.page_size;
|
||||
if (arch.translate(aspace, va)) |phys| {
|
||||
arch.unmapUserPageInto(aspace, va);
|
||||
pmm.free(phys);
|
||||
}
|
||||
}
|
||||
check("munmap unmapped every grant", arch.translate(aspace, arena) == null);
|
||||
arch.destroyAddressSpace(aspace);
|
||||
|
||||
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -761,28 +801,28 @@ fn processTest(boot_info: *const BootInfo) void {
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||
|
||||
usermode.write_count = 0;
|
||||
usermode.write_from_user = false;
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
proc_worker_run = true;
|
||||
proc_worker_ran = false;
|
||||
sched.spawn(procWorker, 4); // kernel task at the processes' priority
|
||||
|
||||
var spawned: u32 = 0;
|
||||
if (usermode.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
||||
if (usermode.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
||||
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
||||
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
||||
|
||||
// Wait (real time) for several heartbeats across the two processes. Each
|
||||
// process sleeps ~1 s between beats, so a few seconds yields several.
|
||||
sched.setPriority(1); // drop below the workers so they get the cores
|
||||
const deadline = arch.millis() + 8000;
|
||||
while (usermode.write_count < 4 and arch.millis() < deadline) sched.yield();
|
||||
while (process.write_count < 4 and arch.millis() < deadline) sched.yield();
|
||||
sched.setPriority(4);
|
||||
proc_worker_run = false;
|
||||
|
||||
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, usermode.write_count });
|
||||
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, process.write_count });
|
||||
check("both init processes spawned on their own address spaces", spawned == 2);
|
||||
check("processes made repeated heartbeat syscalls (>=4)", usermode.write_count >= 4);
|
||||
check("heartbeats came from user mode (CPL 3)", usermode.write_from_user);
|
||||
check("processes made repeated heartbeat syscalls (>=4)", process.write_count >= 4);
|
||||
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
||||
check("a kernel task coexisted with the processes (preemption)", proc_worker_ran);
|
||||
result();
|
||||
}
|
||||
@@ -794,7 +834,7 @@ fn processTest(boot_info: *const BootInfo) void {
|
||||
fn userPfTest() void {
|
||||
log("DANOS-TEST-BEGIN: user-pf\n", .{});
|
||||
sched.setPreemption(false);
|
||||
_ = usermode.run(usermode.pfBlob()) catch {};
|
||||
_ = process.run(process.pfBlob()) catch {};
|
||||
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
||||
}
|
||||
|
||||
@@ -811,8 +851,8 @@ fn initTest(boot_info: *const BootInfo) void {
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||
usermode.write_count = 0;
|
||||
const spawned = if (usermode.spawnProcess(image, 4)) true else |err| blk: {
|
||||
process.write_count = 0;
|
||||
const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: {
|
||||
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
||||
break :blk false;
|
||||
};
|
||||
@@ -822,15 +862,15 @@ fn initTest(boot_info: *const BootInfo) void {
|
||||
// and sleeps repeatedly (init sleeps ~1 s between beats).
|
||||
sched.setPriority(1);
|
||||
const deadline = arch.millis() + 8000;
|
||||
while (usermode.write_count < 2 and arch.millis() < deadline) sched.yield();
|
||||
while (process.write_count < 2 and arch.millis() < deadline) sched.yield();
|
||||
sched.setPriority(4);
|
||||
|
||||
const prefix = "init: heartbeat";
|
||||
const beat_ok = usermode.write_len >= prefix.len and eql(usermode.write_buf[0..prefix.len], prefix);
|
||||
check("init produced repeated heartbeats (>=2)", usermode.write_count >= 2);
|
||||
const beat_ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix);
|
||||
check("init produced repeated heartbeats (>=2)", process.write_count >= 2);
|
||||
check("heartbeat text arrived intact", beat_ok);
|
||||
check("heartbeats came from user mode (CPL 3)", usermode.write_from_user);
|
||||
check("init is still alive (did not exit)", usermode.exit_code == 0);
|
||||
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
||||
check("init is still alive (did not exit)", process.exit_code == 0);
|
||||
result();
|
||||
}
|
||||
|
||||
|
||||
@@ -58,6 +58,26 @@ pub const page_size = 4096;
|
||||
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||
|
||||
/// The kernel syscall numbers — the single source of truth shared by the kernel
|
||||
/// dispatcher (src/kernel/process.zig) and the user runtime library, so the two
|
||||
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||
pub const Syscall = enum(u64) {
|
||||
exit = 0, // exit(code): end the calling process
|
||||
yield = 1, // yield(): give up the rest of this quantum
|
||||
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
_,
|
||||
};
|
||||
|
||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||
pub const prot_read: u64 = 1;
|
||||
pub const prot_write: u64 = 2;
|
||||
pub const prot_exec: u64 = 4;
|
||||
|
||||
/// Physical address -> its virtual address in the physmap. The single way the
|
||||
/// kernel dereferences a physical address once paging is up.
|
||||
///
|
||||
|
||||
+3
-3
@@ -150,9 +150,9 @@ CASES = [
|
||||
"expect": r"page fault \(vector 14\)",
|
||||
"fail": r"NX not enforced"},
|
||||
{"name": "fault-null", "expect": r"page fault \(vector 14\)"},
|
||||
# Ring 3: a user program runs at CPL 3, makes int 0x80 syscalls, survives
|
||||
# timer interrupts, and exits back into the kernel.
|
||||
{"name": "user",
|
||||
# Memory grants: the mmap/munmap path hands out user pages into a process's
|
||||
# arena, translate resolves them, munmap frees them, and no frames leak.
|
||||
{"name": "usermem",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
||||
|
||||
Reference in New Issue
Block a user