M3: real user processes — address spaces, syscall/sysret, swapgs
Per-process address spaces (AddressSpace = a PML4 with an empty user half and the shared kernel half copied in; create/destroy in paging.zig) with CR3 switched on context switch and TSS.rsp0/kernel_rsp published per switch. The GS base now points at an arch per-CPU block and every ring transition observes the swapgs discipline, so a ring-3 `mov %ax,%gs` can no longer poison per-CPU access. syscall/sysret is the primary user entry (int 0x80 kept as a test path); one handler, installed once at boot, serves both and dispatches on whether the caller is a scheduled process or a borrowed test thread. spawnProcess loads an ELF into a fresh address space and schedules it; exit frees the address space after switching to the kernel tables. New `process` test: init runs twice as a real process (create/exit/recreate) on its own page tables, coexisting with a kernel task under preemption. Suite 28/28. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
5d57e7b01c
commit
37fb3cb0cf
@@ -44,6 +44,8 @@ const Task = struct {
|
||||
// Physical PML4 of this task's address space, or 0 for a kernel task (which
|
||||
// runs on the shared kernel page tables). A user task carries its own.
|
||||
pml4: u64 = 0,
|
||||
user_rip: u64 = 0, // ring-3 entry point (user task only)
|
||||
user_rsp: u64 = 0, // ring-3 stack pointer (user task only)
|
||||
next: ?*Task = null, // ready-queue link
|
||||
};
|
||||
|
||||
@@ -228,6 +230,45 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
return ok;
|
||||
}
|
||||
|
||||
/// Spawn a **user** task: a task with its own address space (`pml4`) that starts
|
||||
/// in ring 3 at `entry_rip` on `user_rsp`. It gets a fresh kernel stack for
|
||||
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
||||
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `pml4` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Priority) bool {
|
||||
const t = freeSlot() orelse return false;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
.priority = priority,
|
||||
.stack = stack,
|
||||
.pml4 = pml4,
|
||||
.user_rip = entry_rip,
|
||||
.user_rsp = user_rsp,
|
||||
};
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.kstack_top = top;
|
||||
// First switch-in lands in startUserTask (no register smuggling — it reads
|
||||
// the ring-3 entry/stack from the Task itself).
|
||||
t.rsp = arch.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||
enqueue(t);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
|
||||
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
|
||||
/// Task avoids smuggling values through callee-saved registers across the
|
||||
/// context switch and lock release.
|
||||
fn startUserTask() void {
|
||||
const t = cur();
|
||||
var buf: [96]u8 = undefined;
|
||||
arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask rip=0x{x} rsp=0x{x} pml4=0x{x} kstack=0x{x}\n", .{ t.user_rip, t.user_rsp, t.pml4, t.kstack_top }) catch "");
|
||||
arch.jumpToUser(t.user_rip, t.user_rsp); // noreturn
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
||||
/// Returns the new task so a core can keep a handle to its idle task.
|
||||
@@ -428,6 +469,37 @@ pub fn exit() noreturn {
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// End the current **user** task: free its address space, then exit. Runs on the
|
||||
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
|
||||
/// switch to the kernel tables that must happen before we free the process's own
|
||||
/// tables — we can't free the page tables we're standing on). The kernel stack
|
||||
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
|
||||
pub fn exitUser() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
const dying = pc.current;
|
||||
const as = dying.pml4;
|
||||
if (as != 0) {
|
||||
const kpml4 = arch.kernelPageTable();
|
||||
arch.loadPageTable(kpml4); // off the process tables before freeing them
|
||||
pc.loaded_pml4 = kpml4;
|
||||
arch.destroyAddressSpace(as);
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.pml4 = 0;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
var discard: usize = 0;
|
||||
switchTo(pc, &discard, next);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Whether the running task is a user process (has its own address space).
|
||||
pub fn currentIsUserProcess() bool {
|
||||
return cur().pml4 != 0;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
return cur().id;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user