M3: real user processes — address spaces, syscall/sysret, swapgs
Per-process address spaces (AddressSpace = a PML4 with an empty user half and the shared kernel half copied in; create/destroy in paging.zig) with CR3 switched on context switch and TSS.rsp0/kernel_rsp published per switch. The GS base now points at an arch per-CPU block and every ring transition observes the swapgs discipline, so a ring-3 `mov %ax,%gs` can no longer poison per-CPU access. syscall/sysret is the primary user entry (int 0x80 kept as a test path); one handler, installed once at boot, serves both and dispatches on whether the caller is a scheduled process or a borrowed test thread. spawnProcess loads an ELF into a fresh address space and schedules it; exit frees the address space after switching to the kernel tables. New `process` test: init runs twice as a real process (create/exit/recreate) on its own page tables, coexisting with a kernel task under preemption. Suite 28/28. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
5d57e7b01c
commit
37fb3cb0cf
+9
-5
@@ -1,11 +1,15 @@
|
|||||||
# System Calls
|
# System Calls
|
||||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||||
|
|
||||||
> **Status:** danos currently has a placeholder M1 surface behind an `int 0x80`
|
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||||
> gate — `0 = exit(code)`, `1 = ping(value)`, `2 = write(ptr, len)` (see
|
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||||
> `src/kernel/usermode.zig`, used by `sbin/init.zig`). It exists to prove the
|
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||||
> ring transition; the microkernel set below replaces it (via `syscall`/`sysret`)
|
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||||
> when processes land (M3).
|
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||||
|
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/usermode.zig`); the
|
||||||
|
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||||
|
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||||
|
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||||
|
|
||||||
## The Mechanism of a Syscall
|
## The Mechanism of a Syscall
|
||||||
|
|
||||||
|
|||||||
@@ -64,8 +64,25 @@ pub fn init() void {
|
|||||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||||
/// kernel's segment layout). Call once the frame allocator is up.
|
/// kernel's segment layout). Call once the frame allocator is up.
|
||||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void {
|
||||||
paging.init(allocFrame, boot_info);
|
paging.init(allocFrame, freeFrame, boot_info);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create a new address space (returns its physical PML4, or null). Shares the
|
||||||
|
/// kernel's higher half; the user (low) half starts empty.
|
||||||
|
pub fn createAddressSpace() ?u64 {
|
||||||
|
return paging.createAddressSpace();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Free an address space and everything mapped in its user half. Caller must not
|
||||||
|
/// be running on it.
|
||||||
|
pub fn destroyAddressSpace(pml4: u64) void {
|
||||||
|
paging.destroyAddressSpace(pml4);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a ring-3 page into address space `pml4` (W^X is the caller's contract).
|
||||||
|
pub fn mapUserPageInto(pml4: u64, virt: u64, phys: u64, writable: bool, executable: bool) void {
|
||||||
|
paging.mapUserInto(pml4, virt, phys, writable, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||||
@@ -369,6 +386,17 @@ pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
|||||||
return sp;
|
return sp;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Drop the current (kernel-context) task to ring 3 at `rip` on `rsp`, never
|
||||||
|
/// returning (defined in isr.s). Used by the scheduler's user-task trampoline
|
||||||
|
/// once it has switched onto the task and read its entry/stack. Interrupts are
|
||||||
|
/// disabled across the swapgs+iretq so no interrupt observes the user GS base in
|
||||||
|
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
|
||||||
|
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
|
||||||
|
|
||||||
|
pub fn jumpToUser(rip: u64, rsp: u64) noreturn {
|
||||||
|
jump_to_user(rip, rsp);
|
||||||
|
}
|
||||||
|
|
||||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||||
/// return. Until set, faults just halt the core.
|
/// return. Until set, faults just halt the core.
|
||||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||||
|
|||||||
@@ -98,6 +98,31 @@ task_trampoline:
|
|||||||
1: hlt # if the entry returns, idle (still preemptible)
|
1: hlt # if the entry returns, idle (still preemptible)
|
||||||
jmp 1b
|
jmp 1b
|
||||||
|
|
||||||
|
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
|
||||||
|
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
|
||||||
|
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
|
||||||
|
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
|
||||||
|
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
|
||||||
|
# loaded this task's address space (CR3) and published its kernel stack
|
||||||
|
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
|
||||||
|
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
|
||||||
|
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
|
||||||
|
# to ring 3, never returning. The scheduler calls this from a fresh user task's
|
||||||
|
# trampoline (after the lock is released and the entry/stack read from the Task).
|
||||||
|
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
|
||||||
|
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
|
||||||
|
# re-enables interrupts on the drop to ring 3.
|
||||||
|
.global jump_to_user
|
||||||
|
jump_to_user:
|
||||||
|
cli
|
||||||
|
push $0x1B # user SS (0x18 | RPL 3)
|
||||||
|
push %rsi # user RSP
|
||||||
|
push $0x202 # RFLAGS: IF | reserved-1
|
||||||
|
push $0x23 # user CS (0x20 | RPL 3)
|
||||||
|
push %rdi # user RIP
|
||||||
|
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||||
|
iretq
|
||||||
|
|
||||||
# --- ring 3 entry/exit ------------------------------------------------------
|
# --- ring 3 entry/exit ------------------------------------------------------
|
||||||
|
|
||||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||||
|
|||||||
@@ -29,6 +29,7 @@ const pf_w: u32 = 2;
|
|||||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
||||||
var kernel_pml4: u64 = 0;
|
var kernel_pml4: u64 = 0;
|
||||||
var alloc_frame: *const fn () ?u64 = undefined;
|
var alloc_frame: *const fn () ?u64 = undefined;
|
||||||
|
var free_frame: *const fn (u64) void = undefined; // for tearing down address spaces
|
||||||
|
|
||||||
/// Set once the kernel is running on its own tables (past the CR3 load in
|
/// Set once the kernel is running on its own tables (past the CR3 load in
|
||||||
/// `init`). Before that, the kernel reaches page-table frames through the
|
/// `init`). Before that, the kernel reaches page-table frames through the
|
||||||
@@ -114,8 +115,9 @@ fn enableNx() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
/// Build the address space and switch onto it.
|
||||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void {
|
||||||
alloc_frame = allocFrame;
|
alloc_frame = allocFrame;
|
||||||
|
free_frame = freeFrame;
|
||||||
enableNx();
|
enableNx();
|
||||||
const pml4 = allocTable();
|
const pml4 = allocTable();
|
||||||
|
|
||||||
@@ -223,10 +225,16 @@ fn descendUser(entry: *u64) u64 {
|
|||||||
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
|
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
|
||||||
/// `descendUser`).
|
/// `descendUser`).
|
||||||
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
|
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
|
||||||
|
mapUserInto(kernel_pml4, virt, phys, writable_page, executable);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a ring-3-accessible page into the address space rooted at `pml4` (which
|
||||||
|
/// may be a process's own table or the kernel's). W^X is the caller's contract.
|
||||||
|
pub fn mapUserInto(pml4: u64, virt: u64, phys: u64, writable_page: bool, executable: bool) void {
|
||||||
var flags: u64 = present | user;
|
var flags: u64 = present | user;
|
||||||
if (writable_page) flags |= writable;
|
if (writable_page) flags |= writable;
|
||||||
if (!executable) flags |= no_execute;
|
if (!executable) flags |= no_execute;
|
||||||
const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
||||||
const pdpt = descendUser(pml4e);
|
const pdpt = descendUser(pml4e);
|
||||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
||||||
const pd = descendUser(pdpte);
|
const pd = descendUser(pdpte);
|
||||||
@@ -236,6 +244,41 @@ pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void
|
|||||||
invalidate(virt);
|
invalidate(virt);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Create a new address space: a fresh PML4 with an empty user half and the
|
||||||
|
/// kernel's higher half shared in (copying PML4[256..512), whose entries point
|
||||||
|
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
|
||||||
|
/// the kernel half propagates to every address space). Returns the physical
|
||||||
|
/// PML4, or null if out of frames.
|
||||||
|
pub fn createAddressSpace() ?u64 {
|
||||||
|
const pml4 = alloc_frame() orelse return null;
|
||||||
|
const t = tableAt(pml4);
|
||||||
|
@memset(t[0..256], 0); // empty user half
|
||||||
|
@memcpy(t[256..512], tableAt(kernel_pml4)[256..512]); // shared kernel half
|
||||||
|
return pml4;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Tear down an address space created by `createAddressSpace`: free every frame
|
||||||
|
/// and table in the user half [0..256), then the PML4 itself. The shared kernel
|
||||||
|
/// half [256..512) is never touched. The caller must not be running on `pml4`.
|
||||||
|
pub fn destroyAddressSpace(pml4: u64) void {
|
||||||
|
const t = tableAt(pml4);
|
||||||
|
for (0..256) |i| {
|
||||||
|
if (t[i] & present != 0) freeSubtree(t[i] & addr_mask, 3); // PDPT level
|
||||||
|
}
|
||||||
|
free_frame(pml4);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At
|
||||||
|
/// level 1 the entries are leaf data frames; above, they are child tables.
|
||||||
|
fn freeSubtree(phys: u64, level: u32) void {
|
||||||
|
const t = tableAt(phys);
|
||||||
|
for (t) |e| {
|
||||||
|
if (e & present == 0) continue;
|
||||||
|
if (level > 1) freeSubtree(e & addr_mask, level - 1) else free_frame(e & addr_mask);
|
||||||
|
}
|
||||||
|
free_frame(phys);
|
||||||
|
}
|
||||||
|
|
||||||
/// Whether `virt` is currently mapped **executable** — present with the NX bit
|
/// Whether `virt` is currently mapped **executable** — present with the NX bit
|
||||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||||
|
|||||||
+5
-1
@@ -128,7 +128,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||||
|
|
||||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||||
arch.enablePaging(pmm.alloc, boot_info);
|
arch.enablePaging(pmm.alloc, pmm.free, boot_info);
|
||||||
log.checkpoint(cp_paging);
|
log.checkpoint(cp_paging);
|
||||||
log.print("\ndanos: paging enabled\n", .{});
|
log.print("\ndanos: paging enabled\n", .{});
|
||||||
log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
||||||
@@ -226,6 +226,10 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
}
|
}
|
||||||
log.checkpoint(cp_discovery);
|
log.checkpoint(cp_discovery);
|
||||||
|
|
||||||
|
// Install the syscall handler (int 0x80 gate + syscall stub) once, before any
|
||||||
|
// user code runs.
|
||||||
|
usermode.init();
|
||||||
|
|
||||||
// Register the current context as the first task before enabling preemption.
|
// Register the current context as the first task before enabling preemption.
|
||||||
scheduler.init(4);
|
scheduler.init(4);
|
||||||
log.checkpoint(cp_scheduler);
|
log.checkpoint(cp_scheduler);
|
||||||
|
|||||||
@@ -44,6 +44,8 @@ const Task = struct {
|
|||||||
// Physical PML4 of this task's address space, or 0 for a kernel task (which
|
// Physical PML4 of this task's address space, or 0 for a kernel task (which
|
||||||
// runs on the shared kernel page tables). A user task carries its own.
|
// runs on the shared kernel page tables). A user task carries its own.
|
||||||
pml4: u64 = 0,
|
pml4: u64 = 0,
|
||||||
|
user_rip: u64 = 0, // ring-3 entry point (user task only)
|
||||||
|
user_rsp: u64 = 0, // ring-3 stack pointer (user task only)
|
||||||
next: ?*Task = null, // ready-queue link
|
next: ?*Task = null, // ready-queue link
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -228,6 +230,45 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
|||||||
return ok;
|
return ok;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Spawn a **user** task: a task with its own address space (`pml4`) that starts
|
||||||
|
/// in ring 3 at `entry_rip` on `user_rsp`. It gets a fresh kernel stack for
|
||||||
|
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
||||||
|
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||||
|
/// **Caller must hold the kernel lock** (the loader that builds `pml4` holds it
|
||||||
|
/// across the whole spawn, so the address space and the task appear atomically).
|
||||||
|
pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Priority) bool {
|
||||||
|
const t = freeSlot() orelse return false;
|
||||||
|
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||||
|
t.* = .{
|
||||||
|
.id = next_id,
|
||||||
|
.state = .ready,
|
||||||
|
.priority = priority,
|
||||||
|
.stack = stack,
|
||||||
|
.pml4 = pml4,
|
||||||
|
.user_rip = entry_rip,
|
||||||
|
.user_rsp = user_rsp,
|
||||||
|
};
|
||||||
|
next_id += 1;
|
||||||
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
|
t.kstack_top = top;
|
||||||
|
// First switch-in lands in startUserTask (no register smuggling — it reads
|
||||||
|
// the ring-3 entry/stack from the Task itself).
|
||||||
|
t.rsp = arch.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||||
|
enqueue(t);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
|
||||||
|
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
|
||||||
|
/// Task avoids smuggling values through callee-saved registers across the
|
||||||
|
/// context switch and lock release.
|
||||||
|
fn startUserTask() void {
|
||||||
|
const t = cur();
|
||||||
|
var buf: [96]u8 = undefined;
|
||||||
|
arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask rip=0x{x} rsp=0x{x} pml4=0x{x} kstack=0x{x}\n", .{ t.user_rip, t.user_rsp, t.pml4, t.kstack_top }) catch "");
|
||||||
|
arch.jumpToUser(t.user_rip, t.user_rsp); // noreturn
|
||||||
|
}
|
||||||
|
|
||||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||||
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
||||||
/// Returns the new task so a core can keep a handle to its idle task.
|
/// Returns the new task so a core can keep a handle to its idle task.
|
||||||
@@ -428,6 +469,37 @@ pub fn exit() noreturn {
|
|||||||
unreachable;
|
unreachable;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// End the current **user** task: free its address space, then exit. Runs on the
|
||||||
|
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
|
||||||
|
/// switch to the kernel tables that must happen before we free the process's own
|
||||||
|
/// tables — we can't free the page tables we're standing on). The kernel stack
|
||||||
|
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
|
||||||
|
pub fn exitUser() noreturn {
|
||||||
|
_ = sync.enter();
|
||||||
|
const pc = thisCpu();
|
||||||
|
const dying = pc.current;
|
||||||
|
const as = dying.pml4;
|
||||||
|
if (as != 0) {
|
||||||
|
const kpml4 = arch.kernelPageTable();
|
||||||
|
arch.loadPageTable(kpml4); // off the process tables before freeing them
|
||||||
|
pc.loaded_pml4 = kpml4;
|
||||||
|
arch.destroyAddressSpace(as);
|
||||||
|
}
|
||||||
|
dying.state = .free;
|
||||||
|
dying.pml4 = 0;
|
||||||
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||||
|
next.state = .running;
|
||||||
|
pc.current = next;
|
||||||
|
var discard: usize = 0;
|
||||||
|
switchTo(pc, &discard, next);
|
||||||
|
unreachable;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the running task is a user process (has its own address space).
|
||||||
|
pub fn currentIsUserProcess() bool {
|
||||||
|
return cur().pml4 != 0;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
pub fn currentId() u32 {
|
||||||
return cur().id;
|
return cur().id;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -99,6 +99,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
|||||||
userPfTest();
|
userPfTest();
|
||||||
} else if (eql(case, "init")) {
|
} else if (eql(case, "init")) {
|
||||||
initTest(boot_info);
|
initTest(boot_info);
|
||||||
|
} else if (eql(case, "process")) {
|
||||||
|
processTest(boot_info);
|
||||||
} else if (eql(case, "poweroff")) {
|
} else if (eql(case, "poweroff")) {
|
||||||
powerTest(.off);
|
powerTest(.off);
|
||||||
} else if (eql(case, "reboot")) {
|
} else if (eql(case, "reboot")) {
|
||||||
@@ -732,6 +734,66 @@ fn userTest() void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var proc_worker_run: bool = true;
|
||||||
|
var proc_worker_ran: bool = false;
|
||||||
|
|
||||||
|
/// A kernel task that spins while a process runs, to prove the two coexist under
|
||||||
|
/// preemption (a process on its own CR3 does not stall kernel work).
|
||||||
|
fn procWorker() void {
|
||||||
|
const running: *volatile bool = &proc_worker_run;
|
||||||
|
const ran: *volatile bool = &proc_worker_ran;
|
||||||
|
while (running.*) ran.* = true;
|
||||||
|
sched.exit();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Real processes: load /sbin/init as a scheduled ring-3 process with its own
|
||||||
|
/// address space, twice in succession. The first run proves a process executes
|
||||||
|
/// on its own page tables (write from CPL 3) and coexists preemptively with a
|
||||||
|
/// kernel task; its exit frees the address space. The second run reuses those
|
||||||
|
/// reclaimed frames — succeeding proves create/exit/teardown/recreate is sound.
|
||||||
|
fn processTest(boot_info: *const BootInfo) void {
|
||||||
|
log("DANOS-TEST-BEGIN: process\n", .{});
|
||||||
|
check("bootloader handed over sbin/init", boot_info.init_len != 0);
|
||||||
|
if (boot_info.init_len == 0) {
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||||
|
const expected = "init: hello from user space\n";
|
||||||
|
|
||||||
|
proc_worker_run = true;
|
||||||
|
proc_worker_ran = false;
|
||||||
|
sched.spawn(procWorker, 4); // kernel task, same priority as the processes
|
||||||
|
|
||||||
|
var runs: u32 = 0;
|
||||||
|
var last_cs: u64 = 0;
|
||||||
|
sched.setPriority(1); // drop below the workers so they get the cores
|
||||||
|
var round: u32 = 0;
|
||||||
|
while (round < 2) : (round += 1) {
|
||||||
|
usermode.write_len = 0;
|
||||||
|
usermode.write_cs = 0;
|
||||||
|
usermode.exit_code = 0xdead;
|
||||||
|
usermode.spawnProcess(image, 4) catch {
|
||||||
|
log("DANOS-PROC: spawn round {d} failed\n", .{round});
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (usermode.exit_code == 0xdead and spins < 5_000_000_000) : (spins += 1) sched.yield();
|
||||||
|
log("DANOS-PROC: round {d} write_len={d} exit_code=0x{x}\n", .{ round, usermode.write_len, usermode.exit_code });
|
||||||
|
if (eql(usermode.write_buf[0..usermode.write_len], expected)) {
|
||||||
|
runs += 1;
|
||||||
|
last_cs = usermode.write_cs;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sched.setPriority(4);
|
||||||
|
proc_worker_run = false;
|
||||||
|
|
||||||
|
check("process ran twice on its own address space (create/exit/recreate)", runs == 2);
|
||||||
|
check("process wrote from CPL 3 (CS = user selector | RPL 3)", last_cs == 0x23);
|
||||||
|
check("a kernel task coexisted with the process (preemption)", proc_worker_ran);
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
|
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
|
||||||
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
|
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
|
||||||
/// RIP. The fault report is the pass signal (matched by the harness); if the
|
/// RIP. The fault report is the pass signal (matched by the harness); if the
|
||||||
|
|||||||
+63
-7
@@ -21,6 +21,7 @@ const danos = @import("danos");
|
|||||||
const arch = @import("arch");
|
const arch = @import("arch");
|
||||||
const pmm = @import("pmm.zig");
|
const pmm = @import("pmm.zig");
|
||||||
const sched = @import("scheduler.zig");
|
const sched = @import("scheduler.zig");
|
||||||
|
const sync = @import("sync.zig");
|
||||||
const log = @import("log.zig");
|
const log = @import("log.zig");
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
const page_size = danos.page_size;
|
||||||
@@ -59,18 +60,33 @@ pub var write_len: usize = 0;
|
|||||||
pub var write_cs: u64 = 0;
|
pub var write_cs: u64 = 0;
|
||||||
pub var exit_code: u64 = 0;
|
pub var exit_code: u64 = 0;
|
||||||
|
|
||||||
/// The M1 syscall surface, dispatched on the saved user rax:
|
/// The M3 syscall surface, dispatched on the saved user rax:
|
||||||
/// 0 = exit(code) — unwind back into the kernel context that entered
|
/// 0 = exit(code) — end the caller (process: free its AS + reschedule;
|
||||||
|
/// borrowed test thread: unwind to the kernel caller)
|
||||||
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
|
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
|
||||||
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
|
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
|
||||||
|
/// 3 = sleep(ms) — block the caller for ms milliseconds
|
||||||
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
|
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
|
||||||
/// replaces this in M3; rax is written back as the return value already, since
|
/// replaces these later; rax is written back as the return value already, since
|
||||||
/// isr_common restores user registers from the trap frame.
|
/// the entry paths restore user registers from the trap frame. One handler
|
||||||
|
/// serves both the int-0x80 gate and the syscall stub.
|
||||||
|
///
|
||||||
|
/// Install it once at boot (before any user code runs) via `init`.
|
||||||
|
pub fn init() void {
|
||||||
|
arch.setSyscallHandler(syscall);
|
||||||
|
}
|
||||||
|
|
||||||
fn syscall(state: *arch.CpuState) void {
|
fn syscall(state: *arch.CpuState) void {
|
||||||
switch (state.rax) {
|
switch (state.rax) {
|
||||||
0 => {
|
0 => {
|
||||||
exit_code = state.rdi;
|
exit_code = state.rdi;
|
||||||
arch.userExit();
|
// A scheduled process frees its address space and reschedules; a
|
||||||
|
// borrowed test thread unwinds back to the kernel that entered it.
|
||||||
|
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
|
||||||
|
},
|
||||||
|
3 => {
|
||||||
|
sched.sleep(state.rdi);
|
||||||
|
state.rax = 0;
|
||||||
},
|
},
|
||||||
1 => {
|
1 => {
|
||||||
if (ping_count < pings.len) {
|
if (ping_count < pings.len) {
|
||||||
@@ -135,7 +151,6 @@ pub fn run(blob: []const u8) RunError!void {
|
|||||||
@memcpy(code[0..blob.len], blob);
|
@memcpy(code[0..blob.len], blob);
|
||||||
@memset(code[blob.len..page_size], 0xCC);
|
@memset(code[blob.len..page_size], 0xCC);
|
||||||
|
|
||||||
arch.setSyscallHandler(syscall);
|
|
||||||
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
|
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
|
||||||
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
|
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
|
||||||
resetRecords();
|
resetRecords();
|
||||||
@@ -272,6 +287,48 @@ fn unloadAll() void {
|
|||||||
loaded_count = 0;
|
loaded_count = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Load one page of a segment into address space `pml4`: a fresh frame, zeroed
|
||||||
|
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||||
|
/// On a later failure the whole address space is torn down, which frees every
|
||||||
|
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||||
|
fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||||
|
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
|
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
||||||
|
@memset(dst[0..page_size], 0);
|
||||||
|
const page_off = page_index * page_size;
|
||||||
|
if (page_off < seg.filesz) {
|
||||||
|
const n = @min(page_size, seg.filesz - page_off);
|
||||||
|
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
|
||||||
|
}
|
||||||
|
arch.mapUserPageInto(pml4, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
||||||
|
/// ring-3 process at `priority`. Unlike `runInitElf` (the borrowed-thread test
|
||||||
|
/// path), this returns immediately — the process runs preemptively on its own
|
||||||
|
/// page tables alongside everything else, and its exit is handled by the syscall
|
||||||
|
/// layer. The whole build (address space + ELF load + task) runs under the
|
||||||
|
/// kernel lock so it appears atomically and can't race pmm/heap on another core.
|
||||||
|
pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
||||||
|
var segs: [max_segments]Segment = undefined;
|
||||||
|
const parsed = try parseSegments(image, &segs);
|
||||||
|
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
|
||||||
|
const pml4 = arch.createAddressSpace() orelse return error.OutOfMemory;
|
||||||
|
errdefer arch.destroyAddressSpace(pml4);
|
||||||
|
|
||||||
|
for (segs[0..parsed.count]) |seg| {
|
||||||
|
for (0..seg.pages()) |i| try loadPageInto(pml4, image, seg, i);
|
||||||
|
}
|
||||||
|
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
|
arch.mapUserPageInto(pml4, stack_virt, stack_frame, true, false); // RW + NX
|
||||||
|
|
||||||
|
if (!sched.spawnUserLocked(pml4, parsed.entry, stack_virt + page_size, priority))
|
||||||
|
return error.OutOfMemory;
|
||||||
|
}
|
||||||
|
|
||||||
/// Load a user ELF image, run it in ring 3 from its entry point, and return its
|
/// Load a user ELF image, run it in ring 3 from its entry point, and return its
|
||||||
/// exit code. Same caller contract as `run` (preemption off, one core). On
|
/// exit code. Same caller contract as `run` (preemption off, one core). On
|
||||||
/// success the user mappings are left in place — teardown comes with real
|
/// success the user mappings are left in place — teardown comes with real
|
||||||
@@ -290,7 +347,6 @@ pub fn runInitElf(image: []const u8) InitError!u64 {
|
|||||||
loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame };
|
loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame };
|
||||||
loaded_count += 1;
|
loaded_count += 1;
|
||||||
|
|
||||||
arch.setSyscallHandler(syscall);
|
|
||||||
resetRecords();
|
resetRecords();
|
||||||
arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size);
|
arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size);
|
||||||
arch.enableInterrupts(); // the exit arrived through an interrupt gate
|
arch.enableInterrupts(); // the exit arrived through an interrupt gate
|
||||||
|
|||||||
@@ -165,6 +165,13 @@ CASES = [
|
|||||||
{"name": "init",
|
{"name": "init",
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
# Real processes: /sbin/init loaded as a scheduled ring-3 process with its
|
||||||
|
# own address space, run twice (create/exit/teardown/recreate), coexisting
|
||||||
|
# with a kernel task under preemption.
|
||||||
|
{"name": "process",
|
||||||
|
"smp": 4,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
||||||
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
||||||
{"name": "poweroff",
|
{"name": "poweroff",
|
||||||
|
|||||||
Reference in New Issue
Block a user