From 37fb3cb0cf48c5b79884985330508b2ee4da913a Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Thu, 9 Jul 2026 00:04:20 +0100 Subject: [PATCH] =?UTF-8?q?M3:=20real=20user=20processes=20=E2=80=94=20add?= =?UTF-8?q?ress=20spaces,=20syscall/sysret,=20swapgs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per-process address spaces (AddressSpace = a PML4 with an empty user half and the shared kernel half copied in; create/destroy in paging.zig) with CR3 switched on context switch and TSS.rsp0/kernel_rsp published per switch. The GS base now points at an arch per-CPU block and every ring transition observes the swapgs discipline, so a ring-3 `mov %ax,%gs` can no longer poison per-CPU access. syscall/sysret is the primary user entry (int 0x80 kept as a test path); one handler, installed once at boot, serves both and dispatches on whether the caller is a scheduled process or a borrowed test thread. spawnProcess loads an ELF into a fresh address space and schedules it; exit frees the address space after switching to the kernel tables. New `process` test: init runs twice as a real process (create/exit/recreate) on its own page tables, coexisting with a kernel task under preemption. Suite 28/28. Co-Authored-By: Claude Fable 5 --- docs/syscall.md | 14 +++--- src/kernel/arch/x86_64/cpu.zig | 32 +++++++++++++- src/kernel/arch/x86_64/isr.s | 25 +++++++++++ src/kernel/arch/x86_64/paging.zig | 47 +++++++++++++++++++- src/kernel/main.zig | 6 ++- src/kernel/scheduler.zig | 72 +++++++++++++++++++++++++++++++ src/kernel/tests.zig | 62 ++++++++++++++++++++++++++ src/kernel/usermode.zig | 70 +++++++++++++++++++++++++++--- test/qemu_test.py | 7 +++ 9 files changed, 318 insertions(+), 17 deletions(-) diff --git a/docs/syscall.md b/docs/syscall.md index eabbf2b..8873928 100644 --- a/docs/syscall.md +++ b/docs/syscall.md @@ -1,11 +1,15 @@ # System Calls System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel). -> **Status:** danos currently has a placeholder M1 surface behind an `int 0x80` -> gate — `0 = exit(code)`, `1 = ping(value)`, `2 = write(ptr, len)` (see -> `src/kernel/usermode.zig`, used by `sbin/init.zig`). It exists to prove the -> ring transition; the microkernel set below replaces it (via `syscall`/`sysret`) -> when processes land (M3). +> **Status:** danos has real user processes (M3). User programs enter the kernel +> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in +> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt +> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The +> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`, +> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/usermode.zig`); the +> handler dispatches on whether the caller is a scheduled process (its own address +> space) or a borrowed test thread. The microkernel set below (IPC_Call / +> IPC_ReplyWait / Yield) replaces it once a second user server exists. ## The Mechanism of a Syscall diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index 6c0b4e1..f5bb3b2 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -64,8 +64,25 @@ pub fn init() void { /// Build the kernel's own page tables (with real permissions) and switch onto /// them. Needs the frame allocator and the boot info (for the memory map and the /// kernel's segment layout). Call once the frame allocator is up. -pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void { - paging.init(allocFrame, boot_info); +pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void { + paging.init(allocFrame, freeFrame, boot_info); +} + +/// Create a new address space (returns its physical PML4, or null). Shares the +/// kernel's higher half; the user (low) half starts empty. +pub fn createAddressSpace() ?u64 { + return paging.createAddressSpace(); +} + +/// Free an address space and everything mapped in its user half. Caller must not +/// be running on it. +pub fn destroyAddressSpace(pml4: u64) void { + paging.destroyAddressSpace(pml4); +} + +/// Map a ring-3 page into address space `pml4` (W^X is the caller's contract). +pub fn mapUserPageInto(pml4: u64, virt: u64, phys: u64, writable: bool, executable: bool) void { + paging.mapUserInto(pml4, virt, phys, writable, executable); } /// Map a page into the kernel address space (non-executable). For the heap, etc. @@ -369,6 +386,17 @@ pub fn initTaskStack(stack_top: usize, entry: usize) usize { return sp; } +/// Drop the current (kernel-context) task to ring 3 at `rip` on `rsp`, never +/// returning (defined in isr.s). Used by the scheduler's user-task trampoline +/// once it has switched onto the task and read its entry/stack. Interrupts are +/// disabled across the swapgs+iretq so no interrupt observes the user GS base in +/// ring 0; the pushed RFLAGS re-enables them in ring 3. +extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn; + +pub fn jumpToUser(rip: u64, rsp: u64) noreturn { + jump_to_user(rip, rsp); +} + /// Route CPU exceptions to `handler`, which receives the trap frame and does not /// return. Until set, faults just halt the core. pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void { diff --git a/src/kernel/arch/x86_64/isr.s b/src/kernel/arch/x86_64/isr.s index 185278e..f4888d6 100644 --- a/src/kernel/arch/x86_64/isr.s +++ b/src/kernel/arch/x86_64/isr.s @@ -98,6 +98,31 @@ task_trampoline: 1: hlt # if the entry returns, idle (still preemptible) jmp 1b +# user_task_trampoline: the first thing a freshly-spawned *user* task runs. +# init_user_task_stack leaves the user entry in r15 and the user stack in r14 +# (both callee-saved, so they survive the lock-release call). Like task_trampoline +# it drops the inherited kernel lock, then — instead of calling a kernel fn — it +# builds an iretq frame and drops to ring 3. The scheduler's switchTo already +# loaded this task's address space (CR3) and published its kernel stack +# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls +# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed). +# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context +# to ring 3, never returning. The scheduler calls this from a fresh user task's +# trampoline (after the lock is released and the entry/stack read from the Task). +# cli guards the swapgs..iretq window: an interrupt there would run in ring 0 +# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set) +# re-enables interrupts on the drop to ring 3. +.global jump_to_user +jump_to_user: + cli + push $0x1B # user SS (0x18 | RPL 3) + push %rsi # user RSP + push $0x202 # RFLAGS: IF | reserved-1 + push $0x23 # user CS (0x20 | RPL 3) + push %rdi # user RIP + swapgs # user GS base (isr_common/syscall swap back on entry) + iretq + # --- ring 3 entry/exit ------------------------------------------------------ # enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0) diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index aa98e80..8b2a271 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -29,6 +29,7 @@ const pf_w: u32 = 2; // State kept after init so map()/unmap() can serve later callers (e.g. the heap). var kernel_pml4: u64 = 0; var alloc_frame: *const fn () ?u64 = undefined; +var free_frame: *const fn (u64) void = undefined; // for tearing down address spaces /// Set once the kernel is running on its own tables (past the CR3 load in /// `init`). Before that, the kernel reaches page-table frames through the @@ -114,8 +115,9 @@ fn enableNx() void { } /// Build the address space and switch onto it. -pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void { +pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_info: *const danos.BootInfo) void { alloc_frame = allocFrame; + free_frame = freeFrame; enableNx(); const pml4 = allocTable(); @@ -223,10 +225,16 @@ fn descendUser(entry: *u64) u64 { /// writable + no-execute. `virt` must lie in a user-exclusive region (see /// `descendUser`). pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void { + mapUserInto(kernel_pml4, virt, phys, writable_page, executable); +} + +/// Map a ring-3-accessible page into the address space rooted at `pml4` (which +/// may be a process's own table or the kernel's). W^X is the caller's contract. +pub fn mapUserInto(pml4: u64, virt: u64, phys: u64, writable_page: bool, executable: bool) void { var flags: u64 = present | user; if (writable_page) flags |= writable; if (!executable) flags |= no_execute; - const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF]; + const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF]; const pdpt = descendUser(pml4e); const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF]; const pd = descendUser(pdpte); @@ -236,6 +244,41 @@ pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void invalidate(virt); } +/// Create a new address space: a fresh PML4 with an empty user half and the +/// kernel's higher half shared in (copying PML4[256..512), whose entries point +/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in +/// the kernel half propagates to every address space). Returns the physical +/// PML4, or null if out of frames. +pub fn createAddressSpace() ?u64 { + const pml4 = alloc_frame() orelse return null; + const t = tableAt(pml4); + @memset(t[0..256], 0); // empty user half + @memcpy(t[256..512], tableAt(kernel_pml4)[256..512]); // shared kernel half + return pml4; +} + +/// Tear down an address space created by `createAddressSpace`: free every frame +/// and table in the user half [0..256), then the PML4 itself. The shared kernel +/// half [256..512) is never touched. The caller must not be running on `pml4`. +pub fn destroyAddressSpace(pml4: u64) void { + const t = tableAt(pml4); + for (0..256) |i| { + if (t[i] & present != 0) freeSubtree(t[i] & addr_mask, 3); // PDPT level + } + free_frame(pml4); +} + +/// Recursively free a page-table subtree: `level` 3 = PDPT, 2 = PD, 1 = PT. At +/// level 1 the entries are leaf data frames; above, they are child tables. +fn freeSubtree(phys: u64, level: u32) void { + const t = tableAt(phys); + for (t) |e| { + if (e & present == 0) continue; + if (level > 1) freeSubtree(e & addr_mask, level - 1) else free_frame(e & addr_mask); + } + free_frame(phys); +} + /// Whether `virt` is currently mapped **executable** — present with the NX bit /// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page /// case). Returns false if unmapped. Used for W^X checks in tests. diff --git a/src/kernel/main.zig b/src/kernel/main.zig index e93edd8..75e1c02 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -128,7 +128,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); // Switch off the firmware's page tables onto our own (with real permissions). - arch.enablePaging(pmm.alloc, boot_info); + arch.enablePaging(pmm.alloc, pmm.free, boot_info); log.checkpoint(cp_paging); log.print("\ndanos: paging enabled\n", .{}); log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()}); @@ -226,6 +226,10 @@ fn kmain(boot_info: *const BootInfo) noreturn { } log.checkpoint(cp_discovery); + // Install the syscall handler (int 0x80 gate + syscall stub) once, before any + // user code runs. + usermode.init(); + // Register the current context as the first task before enabling preemption. scheduler.init(4); log.checkpoint(cp_scheduler); diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 26dd53b..196b636 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -44,6 +44,8 @@ const Task = struct { // Physical PML4 of this task's address space, or 0 for a kernel task (which // runs on the shared kernel page tables). A user task carries its own. pml4: u64 = 0, + user_rip: u64 = 0, // ring-3 entry point (user task only) + user_rsp: u64 = 0, // ring-3 stack pointer (user task only) next: ?*Task = null, // ready-queue link }; @@ -228,6 +230,45 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { return ok; } +/// Spawn a **user** task: a task with its own address space (`pml4`) that starts +/// in ring 3 at `entry_rip` on `user_rsp`. It gets a fresh kernel stack for +/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`. +/// Returns false (creating nothing) if the table is full or out of memory. +/// **Caller must hold the kernel lock** (the loader that builds `pml4` holds it +/// across the whole spawn, so the address space and the task appear atomically). +pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Priority) bool { + const t = freeSlot() orelse return false; + const stack = heap.allocator().alloc(u8, stack_size) catch return false; + t.* = .{ + .id = next_id, + .state = .ready, + .priority = priority, + .stack = stack, + .pml4 = pml4, + .user_rip = entry_rip, + .user_rsp = user_rsp, + }; + next_id += 1; + const top = @intFromPtr(stack.ptr) + stack.len; + t.kstack_top = top; + // First switch-in lands in startUserTask (no register smuggling — it reads + // the ring-3 entry/stack from the Task itself). + t.rsp = arch.initTaskStack(top, @intFromPtr(&startUserTask)); + enqueue(t); + return true; +} + +/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It +/// drops to ring 3 at the task's recorded entry/stack. Reading them from the +/// Task avoids smuggling values through callee-saved registers across the +/// context switch and lock release. +fn startUserTask() void { + const t = cur(); + var buf: [96]u8 = undefined; + arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask rip=0x{x} rsp=0x{x} pml4=0x{x} kstack=0x{x}\n", .{ t.user_rip, t.user_rsp, t.pml4, t.kstack_top }) catch ""); + arch.jumpToUser(t.user_rip, t.user_rsp); // noreturn +} + /// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the /// single-threaded boot path). `affinity` pins the task to a core (null = any). /// Returns the new task so a core can keep a handle to its idle task. @@ -428,6 +469,37 @@ pub fn exit() noreturn { unreachable; } +/// End the current **user** task: free its address space, then exit. Runs on the +/// dying task's kernel stack (in the shared kernel half, so it survives the CR3 +/// switch to the kernel tables that must happen before we free the process's own +/// tables — we can't free the page tables we're standing on). The kernel stack +/// itself is leaked, as in `exit` (no reaper yet). Never returns. +pub fn exitUser() noreturn { + _ = sync.enter(); + const pc = thisCpu(); + const dying = pc.current; + const as = dying.pml4; + if (as != 0) { + const kpml4 = arch.kernelPageTable(); + arch.loadPageTable(kpml4); // off the process tables before freeing them + pc.loaded_pml4 = kpml4; + arch.destroyAddressSpace(as); + } + dying.state = .free; + dying.pml4 = 0; + const next = dequeueHighest(pc) orelse @panic("sched: no task left to run"); + next.state = .running; + pc.current = next; + var discard: usize = 0; + switchTo(pc, &discard, next); + unreachable; +} + +/// Whether the running task is a user process (has its own address space). +pub fn currentIsUserProcess() bool { + return cur().pml4 != 0; +} + pub fn currentId() u32 { return cur().id; } diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index 557f11f..af68dd6 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -99,6 +99,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { userPfTest(); } else if (eql(case, "init")) { initTest(boot_info); + } else if (eql(case, "process")) { + processTest(boot_info); } else if (eql(case, "poweroff")) { powerTest(.off); } else if (eql(case, "reboot")) { @@ -732,6 +734,66 @@ fn userTest() void { result(); } +var proc_worker_run: bool = true; +var proc_worker_ran: bool = false; + +/// A kernel task that spins while a process runs, to prove the two coexist under +/// preemption (a process on its own CR3 does not stall kernel work). +fn procWorker() void { + const running: *volatile bool = &proc_worker_run; + const ran: *volatile bool = &proc_worker_ran; + while (running.*) ran.* = true; + sched.exit(); +} + +/// Real processes: load /sbin/init as a scheduled ring-3 process with its own +/// address space, twice in succession. The first run proves a process executes +/// on its own page tables (write from CPL 3) and coexists preemptively with a +/// kernel task; its exit frees the address space. The second run reuses those +/// reclaimed frames — succeeding proves create/exit/teardown/recreate is sound. +fn processTest(boot_info: *const BootInfo) void { + log("DANOS-TEST-BEGIN: process\n", .{}); + check("bootloader handed over sbin/init", boot_info.init_len != 0); + if (boot_info.init_len == 0) { + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; + const expected = "init: hello from user space\n"; + + proc_worker_run = true; + proc_worker_ran = false; + sched.spawn(procWorker, 4); // kernel task, same priority as the processes + + var runs: u32 = 0; + var last_cs: u64 = 0; + sched.setPriority(1); // drop below the workers so they get the cores + var round: u32 = 0; + while (round < 2) : (round += 1) { + usermode.write_len = 0; + usermode.write_cs = 0; + usermode.exit_code = 0xdead; + usermode.spawnProcess(image, 4) catch { + log("DANOS-PROC: spawn round {d} failed\n", .{round}); + continue; + }; + var spins: u64 = 0; + while (usermode.exit_code == 0xdead and spins < 5_000_000_000) : (spins += 1) sched.yield(); + log("DANOS-PROC: round {d} write_len={d} exit_code=0x{x}\n", .{ round, usermode.write_len, usermode.exit_code }); + if (eql(usermode.write_buf[0..usermode.write_len], expected)) { + runs += 1; + last_cs = usermode.write_cs; + } + } + sched.setPriority(4); + proc_worker_run = false; + + check("process ran twice on its own address space (create/exit/recreate)", runs == 2); + check("process wrote from CPL 3 (CS = user selector | RPL 3)", last_cs == 0x23); + check("a kernel task coexisted with the process (preemption)", proc_worker_ran); + result(); +} + /// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present, /// supervisor) must page-fault with error code 0x5 (present | user) at the user /// RIP. The fault report is the pass signal (matched by the harness); if the diff --git a/src/kernel/usermode.zig b/src/kernel/usermode.zig index 2971731..e3119cb 100644 --- a/src/kernel/usermode.zig +++ b/src/kernel/usermode.zig @@ -21,6 +21,7 @@ const danos = @import("danos"); const arch = @import("arch"); const pmm = @import("pmm.zig"); const sched = @import("scheduler.zig"); +const sync = @import("sync.zig"); const log = @import("log.zig"); const page_size = danos.page_size; @@ -59,18 +60,33 @@ pub var write_len: usize = 0; pub var write_cs: u64 = 0; pub var exit_code: u64 = 0; -/// The M1 syscall surface, dispatched on the saved user rax: -/// 0 = exit(code) — unwind back into the kernel context that entered +/// The M3 syscall surface, dispatched on the saved user rax: +/// 0 = exit(code) — end the caller (process: free its AS + reschedule; +/// borrowed test thread: unwind to the kernel caller) /// 1 = ping(value) — record rdi + the caller's CS + the current tick count /// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT: +/// 3 = sleep(ms) — block the caller for ms milliseconds /// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md) -/// replaces this in M3; rax is written back as the return value already, since -/// isr_common restores user registers from the trap frame. +/// replaces these later; rax is written back as the return value already, since +/// the entry paths restore user registers from the trap frame. One handler +/// serves both the int-0x80 gate and the syscall stub. +/// +/// Install it once at boot (before any user code runs) via `init`. +pub fn init() void { + arch.setSyscallHandler(syscall); +} + fn syscall(state: *arch.CpuState) void { switch (state.rax) { 0 => { exit_code = state.rdi; - arch.userExit(); + // A scheduled process frees its address space and reschedules; a + // borrowed test thread unwinds back to the kernel that entered it. + if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit(); + }, + 3 => { + sched.sleep(state.rdi); + state.rax = 0; }, 1 => { if (ping_count < pings.len) { @@ -135,7 +151,6 @@ pub fn run(blob: []const u8) RunError!void { @memcpy(code[0..blob.len], blob); @memset(code[blob.len..page_size], 0xCC); - arch.setSyscallHandler(syscall); arch.mapUserPage(code_virt, code_frame, false, true); // RO + X arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX resetRecords(); @@ -272,6 +287,48 @@ fn unloadAll() void { loaded_count = 0; } +/// Load one page of a segment into address space `pml4`: a fresh frame, zeroed +/// and filled through the physmap, mapped user-accessible with the segment's W^X. +/// On a later failure the whole address space is torn down, which frees every +/// frame mapped into it — so no per-page rollback list is needed here. +fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void { + const frame = pmm.alloc() orelse return error.OutOfMemory; + const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame)); + @memset(dst[0..page_size], 0); + const page_off = page_index * page_size; + if (page_off < seg.filesz) { + const n = @min(page_size, seg.filesz - page_off); + @memcpy(dst[0..n], image[seg.off + page_off ..][0..n]); + } + arch.mapUserPageInto(pml4, seg.vaddr + page_off, frame, seg.writable, seg.executable); +} + +/// Load a user ELF image into a fresh address space and spawn it as a scheduled +/// ring-3 process at `priority`. Unlike `runInitElf` (the borrowed-thread test +/// path), this returns immediately — the process runs preemptively on its own +/// page tables alongside everything else, and its exit is handled by the syscall +/// layer. The whole build (address space + ELF load + task) runs under the +/// kernel lock so it appears atomically and can't race pmm/heap on another core. +pub fn spawnProcess(image: []const u8, priority: u3) InitError!void { + var segs: [max_segments]Segment = undefined; + const parsed = try parseSegments(image, &segs); + + const flags = sync.enter(); + defer sync.leave(flags); + + const pml4 = arch.createAddressSpace() orelse return error.OutOfMemory; + errdefer arch.destroyAddressSpace(pml4); + + for (segs[0..parsed.count]) |seg| { + for (0..seg.pages()) |i| try loadPageInto(pml4, image, seg, i); + } + const stack_frame = pmm.alloc() orelse return error.OutOfMemory; + arch.mapUserPageInto(pml4, stack_virt, stack_frame, true, false); // RW + NX + + if (!sched.spawnUserLocked(pml4, parsed.entry, stack_virt + page_size, priority)) + return error.OutOfMemory; +} + /// Load a user ELF image, run it in ring 3 from its entry point, and return its /// exit code. Same caller contract as `run` (preemption off, one core). On /// success the user mappings are left in place — teardown comes with real @@ -290,7 +347,6 @@ pub fn runInitElf(image: []const u8) InitError!u64 { loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame }; loaded_count += 1; - arch.setSyscallHandler(syscall); resetRecords(); arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size); arch.enableInterrupts(); // the exit arrived through an interrupt gate diff --git a/test/qemu_test.py b/test/qemu_test.py index dd991d3..7d2918f 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -165,6 +165,13 @@ CASES = [ {"name": "init", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Real processes: /sbin/init loaded as a scheduled ring-3 process with its + # own address space, run twice (create/exit/teardown/recreate), coexisting + # with a kernel task under preemption. + {"name": "process", + "smp": 4, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the # pre-transition marker; the FAIL line only appears if the transition didn't take. {"name": "poweroff",