M3: real user processes — address spaces, syscall/sysret, swapgs

Per-process address spaces (AddressSpace = a PML4 with an empty user
half and the shared kernel half copied in; create/destroy in paging.zig)
with CR3 switched on context switch and TSS.rsp0/kernel_rsp published per
switch. The GS base now points at an arch per-CPU block and every ring
transition observes the swapgs discipline, so a ring-3 `mov %ax,%gs` can
no longer poison per-CPU access. syscall/sysret is the primary user entry
(int 0x80 kept as a test path); one handler, installed once at boot,
serves both and dispatches on whether the caller is a scheduled process
or a borrowed test thread. spawnProcess loads an ELF into a fresh address
space and schedules it; exit frees the address space after switching to
the kernel tables. New `process` test: init runs twice as a real process
(create/exit/recreate) on its own page tables, coexisting with a kernel
task under preemption. Suite 28/28.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-09 00:04:20 +01:00
co-authored by Claude Fable 5
parent 5d57e7b01c
commit 37fb3cb0cf
9 changed files with 318 additions and 17 deletions
+30 -2
View File
@@ -64,8 +64,25 @@ pub fn init() void {
/// Build the kernel's own page tables (with real permissions) and switch onto
/// them. Needs the frame allocator and the boot info (for the memory map and the
/// kernel's segment layout). Call once the frame allocator is up.
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
paging.init(allocFrame, boot_info);
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void {
paging.init(allocFrame, freeFrame, boot_info);
}
/// Create a new address space (returns its physical PML4, or null). Shares the
/// kernel's higher half; the user (low) half starts empty.
pub fn createAddressSpace() ?u64 {
return paging.createAddressSpace();
}
/// Free an address space and everything mapped in its user half. Caller must not
/// be running on it.
pub fn destroyAddressSpace(pml4: u64) void {
paging.destroyAddressSpace(pml4);
}
/// Map a ring-3 page into address space `pml4` (W^X is the caller's contract).
pub fn mapUserPageInto(pml4: u64, virt: u64, phys: u64, writable: bool, executable: bool) void {
paging.mapUserInto(pml4, virt, phys, writable, executable);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc.
@@ -369,6 +386,17 @@ pub fn initTaskStack(stack_top: usize, entry: usize) usize {
return sp;
}
/// Drop the current (kernel-context) task to ring 3 at `rip` on `rsp`, never
/// returning (defined in isr.s). Used by the scheduler's user-task trampoline
/// once it has switched onto the task and read its entry/stack. Interrupts are
/// disabled across the swapgs+iretq so no interrupt observes the user GS base in
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
pub fn jumpToUser(rip: u64, rsp: u64) noreturn {
jump_to_user(rip, rsp);
}
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
/// return. Until set, faults just halt the core.
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
+25
View File
@@ -98,6 +98,31 @@ task_trampoline:
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
# loaded this task's address space (CR3) and published its kernel stack
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
# to ring 3, never returning. The scheduler calls this from a fresh user task's
# trampoline (after the lock is released and the entry/stack read from the Task).
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
# re-enables interrupts on the drop to ring 3.
.global jump_to_user
jump_to_user:
cli
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base (isr_common/syscall swap back on entry)
iretq
# --- ring 3 entry/exit ------------------------------------------------------
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
+45 -2
View File
@@ -29,6 +29,7 @@ const pf_w: u32 = 2;
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
var kernel_pml4: u64 = 0;
var alloc_frame: *const fn () ?u64 = undefined;
var free_frame: *const fn (u64) void = undefined; // for tearing down address spaces
/// Set once the kernel is running on its own tables (past the CR3 load in
/// `init`). Before that, the kernel reaches page-table frames through the
@@ -114,8 +115,9 @@ fn enableNx() void {
}
/// Build the address space and switch onto it.
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void {
alloc_frame = allocFrame;
free_frame = freeFrame;
enableNx();
const pml4 = allocTable();
@@ -223,10 +225,16 @@ fn descendUser(entry: *u64) u64 {
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
/// `descendUser`).
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
mapUserInto(kernel_pml4, virt, phys, writable_page, executable);
}
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
/// may be a process's own table or the kernel's). W^X is the caller's contract.
pub fn mapUserInto(pml4: u64, virt: u64, phys: u64, writable_page: bool, executable: bool) void {
var flags: u64 = present | user;
if (writable_page) flags |= writable;
if (!executable) flags |= no_execute;
const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
const pd = descendUser(pdpte);
@@ -236,6 +244,41 @@ pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void
invalidate(virt);
}
/// Create a new address space: a fresh PML4 with an empty user half and the
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
/// the kernel half propagates to every address space). Returns the physical
/// PML4, or null if out of frames.
pub fn createAddressSpace() ?u64 {
const pml4 = alloc_frame() orelse return null;
const t = tableAt(pml4);
@memset(t[0..256], 0); // empty user half
@memcpy(t[256..512], tableAt(kernel_pml4)[256..512]); // shared kernel half
return pml4;
}
/// Tear down an address space created by `createAddressSpace`: free every frame
/// and table in the user half [0..256), then the PML4 itself. The shared kernel
/// half [256..512) is never touched. The caller must not be running on `pml4`.
pub fn destroyAddressSpace(pml4: u64) void {
const t = tableAt(pml4);
for (0..256) |i| {
if (t[i] & present != 0) freeSubtree(t[i] & addr_mask, 3); // PDPT level
}
free_frame(pml4);
}
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
/// level 1 the entries are leaf data frames; above, they are child tables.
fn freeSubtree(phys: u64, level: u32) void {
const t = tableAt(phys);
for (t) |e| {
if (e & present == 0) continue;
if (level > 1) freeSubtree(e & addr_mask, level - 1) else free_frame(e & addr_mask);
}
free_frame(phys);
}
/// Whether `virt` is currently mapped **executable** — present with the NX bit
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
/// case). Returns false if unmapped. Used for W^X checks in tests.
+5 -1
View File
@@ -128,7 +128,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
// Switch off the firmware's page tables onto our own (with real permissions).
arch.enablePaging(pmm.alloc, boot_info);
arch.enablePaging(pmm.alloc, pmm.free, boot_info);
log.checkpoint(cp_paging);
log.print("\ndanos: paging enabled\n", .{});
log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
@@ -226,6 +226,10 @@ fn kmain(boot_info: *const BootInfo) noreturn {
}
log.checkpoint(cp_discovery);
// Install the syscall handler (int 0x80 gate + syscall stub) once, before any
// user code runs.
usermode.init();
// Register the current context as the first task before enabling preemption.
scheduler.init(4);
log.checkpoint(cp_scheduler);
+72
View File
@@ -44,6 +44,8 @@ const Task = struct {
// Physical PML4 of this task's address space, or 0 for a kernel task (which
// runs on the shared kernel page tables). A user task carries its own.
pml4: u64 = 0,
user_rip: u64 = 0, // ring-3 entry point (user task only)
user_rsp: u64 = 0, // ring-3 stack pointer (user task only)
next: ?*Task = null, // ready-queue link
};
@@ -228,6 +230,45 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
return ok;
}
/// Spawn a **user** task: a task with its own address space (`pml4`) that starts
/// in ring 3 at `entry_rip` on `user_rsp`. It gets a fresh kernel stack for
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
/// Returns false (creating nothing) if the table is full or out of memory.
/// **Caller must hold the kernel lock** (the loader that builds `pml4` holds it
/// across the whole spawn, so the address space and the task appear atomically).
pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Priority) bool {
const t = freeSlot() orelse return false;
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
t.* = .{
.id = next_id,
.state = .ready,
.priority = priority,
.stack = stack,
.pml4 = pml4,
.user_rip = entry_rip,
.user_rsp = user_rsp,
};
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
t.kstack_top = top;
// First switch-in lands in startUserTask (no register smuggling — it reads
// the ring-3 entry/stack from the Task itself).
t.rsp = arch.initTaskStack(top, @intFromPtr(&startUserTask));
enqueue(t);
return true;
}
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
/// Task avoids smuggling values through callee-saved registers across the
/// context switch and lock release.
fn startUserTask() void {
const t = cur();
var buf: [96]u8 = undefined;
arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask rip=0x{x} rsp=0x{x} pml4=0x{x} kstack=0x{x}\n", .{ t.user_rip, t.user_rsp, t.pml4, t.kstack_top }) catch "");
arch.jumpToUser(t.user_rip, t.user_rsp); // noreturn
}
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
/// Returns the new task so a core can keep a handle to its idle task.
@@ -428,6 +469,37 @@ pub fn exit() noreturn {
unreachable;
}
/// End the current **user** task: free its address space, then exit. Runs on the
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
/// switch to the kernel tables that must happen before we free the process's own
/// tables — we can't free the page tables we're standing on). The kernel stack
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
pub fn exitUser() noreturn {
_ = sync.enter();
const pc = thisCpu();
const dying = pc.current;
const as = dying.pml4;
if (as != 0) {
const kpml4 = arch.kernelPageTable();
arch.loadPageTable(kpml4); // off the process tables before freeing them
pc.loaded_pml4 = kpml4;
arch.destroyAddressSpace(as);
}
dying.state = .free;
dying.pml4 = 0;
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running;
pc.current = next;
var discard: usize = 0;
switchTo(pc, &discard, next);
unreachable;
}
/// Whether the running task is a user process (has its own address space).
pub fn currentIsUserProcess() bool {
return cur().pml4 != 0;
}
pub fn currentId() u32 {
return cur().id;
}
+62
View File
@@ -99,6 +99,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
userPfTest();
} else if (eql(case, "init")) {
initTest(boot_info);
} else if (eql(case, "process")) {
processTest(boot_info);
} else if (eql(case, "poweroff")) {
powerTest(.off);
} else if (eql(case, "reboot")) {
@@ -732,6 +734,66 @@ fn userTest() void {
result();
}
var proc_worker_run: bool = true;
var proc_worker_ran: bool = false;
/// A kernel task that spins while a process runs, to prove the two coexist under
/// preemption (a process on its own CR3 does not stall kernel work).
fn procWorker() void {
const running: *volatile bool = &proc_worker_run;
const ran: *volatile bool = &proc_worker_ran;
while (running.*) ran.* = true;
sched.exit();
}
/// Real processes: load /sbin/init as a scheduled ring-3 process with its own
/// address space, twice in succession. The first run proves a process executes
/// on its own page tables (write from CPL 3) and coexists preemptively with a
/// kernel task; its exit frees the address space. The second run reuses those
/// reclaimed frames — succeeding proves create/exit/teardown/recreate is sound.
fn processTest(boot_info: *const BootInfo) void {
log("DANOS-TEST-BEGIN: process\n", .{});
check("bootloader handed over sbin/init", boot_info.init_len != 0);
if (boot_info.init_len == 0) {
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
const expected = "init: hello from user space\n";
proc_worker_run = true;
proc_worker_ran = false;
sched.spawn(procWorker, 4); // kernel task, same priority as the processes
var runs: u32 = 0;
var last_cs: u64 = 0;
sched.setPriority(1); // drop below the workers so they get the cores
var round: u32 = 0;
while (round < 2) : (round += 1) {
usermode.write_len = 0;
usermode.write_cs = 0;
usermode.exit_code = 0xdead;
usermode.spawnProcess(image, 4) catch {
log("DANOS-PROC: spawn round {d} failed\n", .{round});
continue;
};
var spins: u64 = 0;
while (usermode.exit_code == 0xdead and spins < 5_000_000_000) : (spins += 1) sched.yield();
log("DANOS-PROC: round {d} write_len={d} exit_code=0x{x}\n", .{ round, usermode.write_len, usermode.exit_code });
if (eql(usermode.write_buf[0..usermode.write_len], expected)) {
runs += 1;
last_cs = usermode.write_cs;
}
}
sched.setPriority(4);
proc_worker_run = false;
check("process ran twice on its own address space (create/exit/recreate)", runs == 2);
check("process wrote from CPL 3 (CS = user selector | RPL 3)", last_cs == 0x23);
check("a kernel task coexisted with the process (preemption)", proc_worker_ran);
result();
}
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
/// RIP. The fault report is the pass signal (matched by the harness); if the
+63 -7
View File
@@ -21,6 +21,7 @@ const danos = @import("danos");
const arch = @import("arch");
const pmm = @import("pmm.zig");
const sched = @import("scheduler.zig");
const sync = @import("sync.zig");
const log = @import("log.zig");
const page_size = danos.page_size;
@@ -59,18 +60,33 @@ pub var write_len: usize = 0;
pub var write_cs: u64 = 0;
pub var exit_code: u64 = 0;
/// The M1 syscall surface, dispatched on the saved user rax:
/// 0 = exit(code) — unwind back into the kernel context that entered
/// The M3 syscall surface, dispatched on the saved user rax:
/// 0 = exit(code) — end the caller (process: free its AS + reschedule;
/// borrowed test thread: unwind to the kernel caller)
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
/// 3 = sleep(ms) — block the caller for ms milliseconds
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
/// replaces this in M3; rax is written back as the return value already, since
/// isr_common restores user registers from the trap frame.
/// replaces these later; rax is written back as the return value already, since
/// the entry paths restore user registers from the trap frame. One handler
/// serves both the int-0x80 gate and the syscall stub.
///
/// Install it once at boot (before any user code runs) via `init`.
pub fn init() void {
arch.setSyscallHandler(syscall);
}
fn syscall(state: *arch.CpuState) void {
switch (state.rax) {
0 => {
exit_code = state.rdi;
arch.userExit();
// A scheduled process frees its address space and reschedules; a
// borrowed test thread unwinds back to the kernel that entered it.
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
},
3 => {
sched.sleep(state.rdi);
state.rax = 0;
},
1 => {
if (ping_count < pings.len) {
@@ -135,7 +151,6 @@ pub fn run(blob: []const u8) RunError!void {
@memcpy(code[0..blob.len], blob);
@memset(code[blob.len..page_size], 0xCC);
arch.setSyscallHandler(syscall);
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
resetRecords();
@@ -272,6 +287,48 @@ fn unloadAll() void {
loaded_count = 0;
}
/// Load one page of a segment into address space `pml4`: a fresh frame, zeroed
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
/// On a later failure the whole address space is torn down, which frees every
/// frame mapped into it — so no per-page rollback list is needed here.
fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
const frame = pmm.alloc() orelse return error.OutOfMemory;
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
@memset(dst[0..page_size], 0);
const page_off = page_index * page_size;
if (page_off < seg.filesz) {
const n = @min(page_size, seg.filesz - page_off);
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
}
arch.mapUserPageInto(pml4, seg.vaddr + page_off, frame, seg.writable, seg.executable);
}
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
/// ring-3 process at `priority`. Unlike `runInitElf` (the borrowed-thread test
/// path), this returns immediately — the process runs preemptively on its own
/// page tables alongside everything else, and its exit is handled by the syscall
/// layer. The whole build (address space + ELF load + task) runs under the
/// kernel lock so it appears atomically and can't race pmm/heap on another core.
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
var segs: [max_segments]Segment = undefined;
const parsed = try parseSegments(image, &segs);
const flags = sync.enter();
defer sync.leave(flags);
const pml4 = arch.createAddressSpace() orelse return error.OutOfMemory;
errdefer arch.destroyAddressSpace(pml4);
for (segs[0..parsed.count]) |seg| {
for (0..seg.pages()) |i| try loadPageInto(pml4, image, seg, i);
}
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
arch.mapUserPageInto(pml4, stack_virt, stack_frame, true, false); // RW + NX
if (!sched.spawnUserLocked(pml4, parsed.entry, stack_virt + page_size, priority))
return error.OutOfMemory;
}
/// Load a user ELF image, run it in ring 3 from its entry point, and return its
/// exit code. Same caller contract as `run` (preemption off, one core). On
/// success the user mappings are left in place — teardown comes with real
@@ -290,7 +347,6 @@ pub fn runInitElf(image: []const u8) InitError!u64 {
loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame };
loaded_count += 1;
arch.setSyscallHandler(syscall);
resetRecords();
arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size);
arch.enableInterrupts(); // the exit arrived through an interrupt gate