diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index f5bb3b2..d75af64 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -20,6 +20,65 @@ const pcpu = @import("percpu.zig"); /// The saved register/trap frame passed to a fault handler. pub const CpuState = idt.CpuState; +// --- trap-frame accessors --------------------------------------------------- +// The frame's fields are x86_64 registers; the generic kernel reads it through +// these accessors so it never names one. + +/// The interrupted/faulting instruction address (RIP here; ELR_EL1 on aarch64, +/// sepc on riscv64). +pub fn instructionPointer(state: *const CpuState) u64 { + return state.rip; +} + +/// The interrupted stack pointer (RSP here). +pub fn stackPointer(state: *const CpuState) u64 { + return state.rsp; +} + +/// Whether the trap came from user mode (CPL 3 here; EL0 on aarch64, U-mode on +/// riscv64). +pub fn fromUser(state: *const CpuState) bool { + return state.cs & 3 == 3; +} + +/// The faulting virtual address, if this trap is a page fault (CR2 here; +/// FAR_EL1 on aarch64, stval on riscv64). Null for any other exception. +pub fn faultAddress(state: *const CpuState) ?u64 { + if (state.vector != 14) return null; + return asm volatile ("mov %%cr2, %[out]" + : [out] "=r" (-> u64), + ); +} + +// --- syscall ABI ------------------------------------------------------------ +// The System V-style register convention (number in rax, arguments in +// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic +// dispatcher never names a register. + +/// The syscall number the user program passed. +pub fn syscallNumber(state: *const CpuState) u64 { + return state.rax; +} + +/// Positional syscall argument `n`. +pub fn syscallArg(state: *const CpuState, n: u8) u64 { + return switch (n) { + 0 => state.rdi, + 1 => state.rsi, + 2 => state.rdx, + 3 => state.r10, + 4 => state.r8, + 5 => state.r9, + else => 0, + }; +} + +/// Write the syscall's return value into the frame — the entry paths restore +/// user registers from it. +pub fn setSyscallResult(state: *CpuState, value: u64) void { + state.rax = value; +} + /// Bring up the serial port (the kernel's machine-readable log). No dependencies, /// so it can be the very first thing called. pub fn serialInit() void { @@ -31,10 +90,11 @@ pub fn serialWrite(bytes: []const u8) void { serial.write(bytes); } -/// Emit a one-byte checkpoint to the POST diagnostic port (0x80). A POST card or -/// BMC displays it; it's the last-resort progress signal when there's no text -/// output at all. Writing 0x80 is universally safe (it's the legacy I/O-delay port). -pub fn postCode(code: u8) void { +/// Emit a one-byte progress checkpoint to whatever hardware debug sink the +/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC +/// displays. The last-resort progress signal when there's no text output at all. +/// Writing 0x80 is universally safe (it's the legacy I/O-delay port). +pub fn checkpoint(code: u8) void { io.outb(0x80, code); } @@ -68,21 +128,22 @@ pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) vo paging.init(allocFrame, freeFrame, boot_info); } -/// Create a new address space (returns its physical PML4, or null). Shares the -/// kernel's higher half; the user (low) half starts empty. +/// Create a new address space (returns the physical address of its root table — +/// the PML4 here — or null). Shares the kernel's higher half; the user (low) +/// half starts empty. pub fn createAddressSpace() ?u64 { return paging.createAddressSpace(); } /// Free an address space and everything mapped in its user half. Caller must not /// be running on it. -pub fn destroyAddressSpace(pml4: u64) void { - paging.destroyAddressSpace(pml4); +pub fn destroyAddressSpace(root: u64) void { + paging.destroyAddressSpace(root); } -/// Map a ring-3 page into address space `pml4` (W^X is the caller's contract). -pub fn mapUserPageInto(pml4: u64, virt: u64, phys: u64, writable: bool, executable: bool) void { - paging.mapUserInto(pml4, virt, phys, writable, executable); +/// Map a user page into address space `root` (W^X is the caller's contract). +pub fn mapUserPageInto(root: u64, virt: u64, phys: u64, writable: bool, executable: bool) void { + paging.mapUserInto(root, virt, phys, writable, executable); } /// Map a page into the kernel address space (non-executable). For the heap, etc. @@ -102,9 +163,17 @@ pub fn kernelPageTable() u64 { return paging.kernelPml4(); } -/// Switch the active address space (load CR3 with a physical PML4). -pub fn loadPageTable(pml4: u64) void { - paging.loadCr3(pml4); +/// Switch the active address space (load CR3 with a physical root table). +pub fn loadPageTable(root: u64) void { + paging.loadCr3(root); +} + +/// The physical root of the currently active page tables (CR3 here; TTBR0/satp +/// elsewhere). +pub fn activePageTable() u64 { + return asm volatile ("mov %%cr3, %[out]" + : [out] "=r" (-> u64), + ); } /// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions: @@ -140,12 +209,12 @@ extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) /// `enter_user` had returned (defined in isr.s). Called by the exit syscall. extern fn user_exit_to_kernel() callconv(.c) noreturn; -/// Run user code at `rip` with stack `rsp` on this core (`cpu` = the caller's CPU -/// index; the arch layer can't ask the scheduler). Returns after the user program -/// exits via syscall. Interrupts are disabled on return (the exit arrives through -/// an interrupt gate) — the caller re-enables. -pub fn enterUser(cpu: usize, rip: u64, rsp: u64) void { - enter_user(rip, rsp, tss.rsp0Ptr(cpu)); +/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the +/// caller's CPU index; the arch layer can't ask the scheduler). Returns after the +/// user program exits via syscall. Interrupts are disabled on return (the exit +/// arrives through an interrupt gate) — the caller re-enables. +pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void { + enter_user(entry, stack_top, tss.rsp0Ptr(cpu)); } /// Never returns to the user program: unwind to the kernel context that called @@ -154,19 +223,12 @@ pub fn userExit() noreturn { user_exit_to_kernel(); } -/// Register the handler for the ring-3 syscall gate (int 0x80, vector 128). The -/// handler may write the trap frame (e.g. rax as the return value). -pub fn setSyscallHandler(handler: *const fn (*idt.CpuState) void) void { +/// Register the handler for the user syscall gate (int 0x80, vector 128). The +/// handler may write the trap frame (see `setSyscallResult`). +pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void { idt.setSyscallHandler(handler); } -/// CR3 holds the physical address of the active top-level page table. -pub fn readCr3() u64 { - return asm volatile ("mov %%cr3, %[out]" - : [out] "=r" (-> u64), - ); -} - /// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each /// core calls this once, after its GDT is in place (a GS *selector* reload would /// clobber the base). See percpu.zig for the swapgs discipline. @@ -190,15 +252,15 @@ pub fn setTrampolinePage(phys: u64) void { smp.setTrampolinePage(phys); } -/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it -/// `stack_top` and its per-CPU pointer `percpu`; it adopts the current (kernel) page -/// tables. Returns false if it doesn't come online within the timeout. Blocks until -/// the core reports in. -pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool { +/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on +/// aarch64, hart id on riscv64) as dense CPU `index`, giving it `stack_top` and +/// its per-CPU pointer `percpu`; it adopts the kernel page tables. Returns false +/// if it doesn't come online within the timeout. Blocks until the core reports in. +pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize) bool { // The AP adopts the kernel page tables explicitly — never the caller's live // CR3, which a future re-wake from a core running a process would make a // process address space. - return smp.startAp(apic_id, stack_top, percpu, index, paging.kernelPml4()); + return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4()); } /// Register the generic entry a woken AP jumps to once its arch state is up (its own @@ -207,11 +269,12 @@ pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void { smp.setSecondaryEntry(entry); } -/// Bytes the kernel should allocate for an AP's IST (double-fault) stack, and where -/// to record its top before waking the core. The stack is heap-allocated per online -/// AP (the BSP's is static — it's needed before the allocator exists). See tss.zig. -pub const ist_stack_size = tss.ist_stack_size; -pub fn setApIstStack(cpu: usize, top: usize) void { +/// Bytes the kernel should allocate for a secondary core's dedicated fault stack +/// (the IST double-fault stack here), and where to record its top before waking +/// the core. The stack is heap-allocated per online core (the boot CPU's is +/// static — it's needed before the allocator exists). See tss.zig. +pub const fault_stack_size = tss.ist_stack_size; +pub fn setFaultStack(cpu: usize, top: usize) void { tss.setApIstStack(cpu, top); } @@ -274,11 +337,12 @@ pub fn timerCalibrationSource() []const u8 { return apic.calibrationSource(); } -/// I/O APIC diagnostics (for boot logging / verification). -pub fn ioapicEntryCount() u32 { +/// External-interrupt-router diagnostics, for boot logging / verification (the +/// I/O APIC's redirection entries here; a GIC distributor or PLIC elsewhere). +pub fn irqRouteCount() u32 { return ioapic.entryCount(); } -pub fn ioapicEntryLow(n: u32) u32 { +pub fn irqRouteRaw(n: u32) u32 { return ioapic.entryLow(n); } @@ -309,11 +373,15 @@ pub fn millis() u64 { return apic.millis(); } -/// Measured LAPIC timer / TSC frequencies in Hz (from calibration). -pub fn lapicHz() u64 { +/// Measured frequency of the tick timer's input clock (the LAPIC timer here), in +/// Hz, from calibration. +pub fn timerClockHz() u64 { return apic.lapicHz(); } -pub fn tscHz() u64 { + +/// Measured frequency of the monotonic clock's underlying counter (the TSC here; +/// CNTVCT on aarch64, `time` on riscv64), in Hz. +pub fn clockHz() u64 { return apic.tscHz(); } @@ -359,8 +427,8 @@ pub fn setTickHook(hook: *const fn () void) void { /// pointer is written to `old_rsp`. Defined in isr.s. extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void; -pub fn switchContext(old_rsp: *usize, new_rsp: usize) void { - switch_context(old_rsp, new_rsp); +pub fn switchContext(old_sp: *usize, new_sp: usize) void { + switch_context(old_sp, new_sp); } /// Build the initial stack for a new task so that switching to it lands in @@ -393,8 +461,8 @@ pub fn initTaskStack(stack_top: usize, entry: usize) usize { /// ring 0; the pushed RFLAGS re-enables them in ring 3. extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn; -pub fn jumpToUser(rip: u64, rsp: u64) noreturn { - jump_to_user(rip, rsp); +pub fn jumpToUser(entry: u64, stack_top: u64) noreturn { + jump_to_user(entry, stack_top); } /// Route CPU exceptions to `handler`, which receives the trap frame and does not @@ -404,7 +472,7 @@ pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void { } /// A human-readable name for a CPU exception vector. -pub fn vectorName(vector: u64) []const u8 { +pub fn exceptionName(vector: u64) []const u8 { return idt.vectorName(vector); } @@ -430,13 +498,6 @@ pub fn pioWrite(width: u8, port: u16, value: u32) void { } } -/// CR2 holds the faulting linear address after a page fault (#PF, vector 14). -pub fn readCr2() u64 { - return asm volatile ("mov %%cr2, %[out]" - : [out] "=r" (-> u64), - ); -} - /// Park the core forever. `hlt` drops it into a low-power idle until the next /// interrupt; the loop re-halts on every wake so the stop is permanent. See /// docs/halting.md for the full reasoning. diff --git a/src/kernel/log.zig b/src/kernel/log.zig index bb2462a..7ca26f2 100644 --- a/src/kernel/log.zig +++ b/src/kernel/log.zig @@ -52,7 +52,7 @@ pub fn print(comptime fmt: []const u8, args: anytype) void { /// progress channel for when there is no text output at all. Independent of the /// sink list, so it works even before any sink is registered. pub fn checkpoint(code: u8) void { - arch.postCode(code); + arch.checkpoint(code); } // --- persistent panic breadcrumb ------------------------------------------- diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 20d0e73..5d539d6 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -131,7 +131,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { arch.enablePaging(pmm.alloc, pmm.free, boot_info); log.checkpoint(cp_paging); log.print("\ndanos: paging enabled\n", .{}); - log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()}); + log.print(" page tables: root = 0x{x:0>16}\n", .{arch.activePageTable()}); log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count}); // Bring up the kernel heap (dynamic allocation), built on the VMM. @@ -216,7 +216,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { } else { log.write(" console UART: none in SPCR -> legacy COM1\n"); } - log.print(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) }); + log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, arch.irqRouteCount(), arch.irqRouteRaw(0) }); const cores = platform.cpus(); log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 }); if (platform.cpusDropped() > 0) @@ -240,7 +240,7 @@ fn kmain(boot_info: *const BootInfo) noreturn { arch.startTimer(); arch.enableInterrupts(); log.checkpoint(cp_timer); - log.print("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() }); + log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.timerClockHz() / 1_000_000, arch.clockHz() / 1_000_000, arch.timerCalibrationSource() }); // Wake the other cores (application processors). A no-op on a single-core // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). @@ -309,13 +309,13 @@ fn bringUpSecondaries() void { continue; }; const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); - // This core's IST (double-fault) stack — allocated only now that the core is + // This core's dedicated fault stack — allocated only now that the core is // real, rather than reserved statically for every possible core. - const ist = heap.allocator().alloc(u8, arch.ist_stack_size) catch { - log.print(" cpu apic_id {d}: no IST stack; skipped\n", .{core.apic_id}); + const fault_stack = heap.allocator().alloc(u8, arch.fault_stack_size) catch { + log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id}); continue; }; - arch.setApIstStack(index, (@intFromPtr(ist.ptr) + ist.len) & ~@as(usize, 15)); + arch.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15)); const pc = scheduler.prepareSecondary(index, core.apic_id); var attempt: u32 = 1; while (attempt <= max_wake_attempts) : (attempt += 1) { @@ -364,14 +364,14 @@ fn onException(state: *const arch.CpuState) noreturn { const core = scheduler.currentCpuIndex(); // A fault is user-facing enough to paint on screen too (via statusPrint), on // top of the diagnostic log. - statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.vectorName(state.vector), state.vector }); + statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.exceptionName(state.vector), state.vector }); statusPrint(" error code : 0x{x}\n", .{state.error_code}); - statusPrint(" RIP : 0x{x:0>16}\n", .{state.rip}); - statusPrint(" RSP : 0x{x:0>16}\n", .{state.rsp}); - if (state.vector == 14) statusPrint(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()}); + statusPrint(" IP : 0x{x:0>16}\n", .{arch.instructionPointer(state)}); + statusPrint(" SP : 0x{x:0>16}\n", .{arch.stackPointer(state)}); + if (arch.faultAddress(state)) |addr| statusPrint(" fault addr : 0x{x:0>16}\n", .{addr}); var buf: [128]u8 = undefined; - log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, core, state.rip }) catch "cpu exception"); + log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ arch.exceptionName(state.vector), state.vector, core, arch.instructionPointer(state) }) catch "cpu exception"); arch.halt(); } diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 196b636..8a6eaf6 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -36,16 +36,16 @@ const Task = struct { id: u32 = 0, state: State = .free, priority: Priority = 0, - rsp: usize = 0, // saved stack pointer, valid while not running + sp: usize = 0, // saved stack pointer, valid while not running stack: []u8 = &.{}, kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core - // Physical PML4 of this task's address space, or 0 for a kernel task (which + // Physical root of this task's address space, or 0 for a kernel task (which // runs on the shared kernel page tables). A user task carries its own. - pml4: u64 = 0, - user_rip: u64 = 0, // ring-3 entry point (user task only) - user_rsp: u64 = 0, // ring-3 stack pointer (user task only) + aspace: u64 = 0, + user_ip: u64 = 0, // user-mode entry point (user task only) + user_sp: u64 = 0, // user-mode stack pointer (user task only) next: ?*Task = null, // ready-queue link }; @@ -66,10 +66,10 @@ var next_id: u32 = 1; pub const PerCpu = struct { current: *Task = undefined, // the task running on this core idle: *Task = undefined, // this core's idle task (always ready, lowest priority) - apic_id: u32 = 0, // the core's Local APIC id + hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64) index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? - loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core + loaded_aspace: u64 = 0, // the address-space root currently loaded on this core // Tasks pinned to this core (affinity == index), per priority level + bitmap. pinned_head: [num_priorities]?*Task = .{null} ** num_priorities, pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities, @@ -105,7 +105,7 @@ var preemption_enabled = true; /// boot, before interrupts are enabled — so no lock is needed here. pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; - pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() }; + pc.* = .{ .index = 0, .online = true, .loaded_aspace = arch.kernelPageTable() }; arch.setCpuLocal(0, @intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; @@ -120,12 +120,12 @@ fn idle() void { } /// Reserve and initialise the per-CPU slot for an application processor at dense -/// `index` (1-based; 0 is the BSP) with Local APIC id `apic_id`, and return a +/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a /// pointer the arch bring-up hands to the core (it publishes it in its GS base). /// Called on the BSP before waking each AP; the AP marks itself `online`. -pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu { +pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu { const pc = &cpus[index]; - pc.* = .{ .index = @intCast(index), .apic_id = apic_id, .online = false }; + pc.* = .{ .index = @intCast(index), .hw_id = hw_id, .online = false }; return pc; } @@ -144,7 +144,7 @@ pub fn secondaryMain() callconv(.c) noreturn { pc.current = t; pc.idle = t; pc.online = true; - pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up + pc.loaded_aspace = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up sync.leave(flags); arch.enableInterrupts(); // the timer now preempts this idle context into work @@ -230,13 +230,13 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { return ok; } -/// Spawn a **user** task: a task with its own address space (`pml4`) that starts -/// in ring 3 at `entry_rip` on `user_rsp`. It gets a fresh kernel stack for +/// Spawn a **user** task: a task with its own address space (`aspace`) that starts +/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for /// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`. /// Returns false (creating nothing) if the table is full or out of memory. -/// **Caller must hold the kernel lock** (the loader that builds `pml4` holds it +/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it /// across the whole spawn, so the address space and the task appear atomically). -pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Priority) bool { +pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool { const t = freeSlot() orelse return false; const stack = heap.allocator().alloc(u8, stack_size) catch return false; t.* = .{ @@ -244,16 +244,16 @@ pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Prior .state = .ready, .priority = priority, .stack = stack, - .pml4 = pml4, - .user_rip = entry_rip, - .user_rsp = user_rsp, + .aspace = aspace, + .user_ip = entry, + .user_sp = user_sp, }; next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; t.kstack_top = top; // First switch-in lands in startUserTask (no register smuggling — it reads - // the ring-3 entry/stack from the Task itself). - t.rsp = arch.initTaskStack(top, @intFromPtr(&startUserTask)); + // the user entry/stack from the Task itself). + t.sp = arch.initTaskStack(top, @intFromPtr(&startUserTask)); enqueue(t); return true; } @@ -265,8 +265,8 @@ pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Prior fn startUserTask() void { const t = cur(); var buf: [96]u8 = undefined; - arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask rip=0x{x} rsp=0x{x} pml4=0x{x} kstack=0x{x}\n", .{ t.user_rip, t.user_rsp, t.pml4, t.kstack_top }) catch ""); - arch.jumpToUser(t.user_rip, t.user_rsp); // noreturn + arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch ""); + arch.jumpToUser(t.user_ip, t.user_sp); // noreturn } /// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the @@ -279,7 +279,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task { next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; t.kstack_top = top; - t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); + t.sp = arch.initTaskStack(top, @intFromPtr(entry)); enqueue(t); return t; } @@ -309,26 +309,26 @@ fn schedule() void { }; next.state = .running; pc.current = next; - if (next != prev) switchTo(pc, &prev.rsp, next); + if (next != prev) switchTo(pc, &prev.sp, next); } /// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a -/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when -/// it differs from what's loaded — every CR3 write is a full TLB flush), then -/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from -/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so -/// this is a no-op beyond the register switch for a pure-kernel workload. The -/// big kernel lock is held and interrupts are off throughout, so no interrupt -/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing -/// task's stack pointer. -fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void { +/// user-mode interrupt lands on a good stack) and its address space (only when +/// it differs from what's loaded — every page-table switch is a full TLB flush), +/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used +/// from user mode) resolve to the shared kernel page tables and skip the kernel- +/// stack write, so this is a no-op beyond the register switch for a pure-kernel +/// workload. The big kernel lock is held and interrupts are off throughout, so no +/// interrupt can observe a half-updated (kernel stack, address space) pair. +/// `save_sp` receives the outgoing task's stack pointer. +fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void { if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top); - const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable(); - if (want != pc.loaded_pml4) { + const want = if (next.aspace != 0) next.aspace else arch.kernelPageTable(); + if (want != pc.loaded_aspace) { arch.loadPageTable(want); - pc.loaded_pml4 = want; + pc.loaded_aspace = want; } - arch.switchContext(save_rsp, next.rsp); + arch.switchContext(save_sp, next.sp); } /// Voluntarily give up the CPU to the next ready task. @@ -478,15 +478,15 @@ pub fn exitUser() noreturn { _ = sync.enter(); const pc = thisCpu(); const dying = pc.current; - const as = dying.pml4; + const as = dying.aspace; if (as != 0) { - const kpml4 = arch.kernelPageTable(); - arch.loadPageTable(kpml4); // off the process tables before freeing them - pc.loaded_pml4 = kpml4; + const kroot = arch.kernelPageTable(); + arch.loadPageTable(kroot); // off the process tables before freeing them + pc.loaded_aspace = kroot; arch.destroyAddressSpace(as); } dying.state = .free; - dying.pml4 = 0; + dying.aspace = 0; const next = dequeueHighest(pc) orelse @panic("sched: no task left to run"); next.state = .running; pc.current = next; @@ -497,7 +497,7 @@ pub fn exitUser() noreturn { /// Whether the running task is a user process (has its own address space). pub fn currentIsUserProcess() bool { - return cur().pml4 != 0; + return cur().aspace != 0; } pub fn currentId() u32 { diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index a7e72c5..fea9d4f 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -166,9 +166,9 @@ fn smoke(boot_info: *const BootInfo) void { if (b) |p| pmm.free(p); check("free returns frames to the pool", pmm.stats().free_frames == before + 2); - // Paging is active on our own tables (CR3 is non-zero and page-aligned). - const cr3 = arch.readCr3(); - check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0); + // Paging is active on our own tables (the root is non-zero and page-aligned). + const root = arch.activePageTable(); + check("paging active (page-table root set)", root != 0 and root % danos.page_size == 0); result(); } @@ -324,10 +324,10 @@ fn heapTest() void { fn clock() void { log("DANOS-TEST-BEGIN: clock\n", .{}); - const lapic = arch.lapicHz(); - check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000); - const tsc = arch.tscHz(); - check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000); + const timer_clock = arch.timerClockHz(); + check("timer clock frequency measured", timer_clock > 1_000_000 and timer_clock < 100_000_000_000); + const clock_hz = arch.clockHz(); + check("monotonic clock frequency measured", clock_hz > 100_000_000 and clock_hz < 100_000_000_000); // Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms). const start_ticks = arch.ticks(); @@ -729,7 +729,7 @@ fn userTest() void { check("user program ran and exited (ring-3 round trip)", ran); check("two ping syscalls received", usermode.ping_count == 2); check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF); - check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23); + check("syscalls came from user mode (CPL 3)", usermode.pings[0].from_user and usermode.pings[1].from_user); check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks); result(); } @@ -762,7 +762,7 @@ fn processTest(boot_info: *const BootInfo) void { const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len]; usermode.write_count = 0; - usermode.write_cs = 0; + usermode.write_from_user = false; proc_worker_run = true; proc_worker_ran = false; sched.spawn(procWorker, 4); // kernel task at the processes' priority @@ -782,7 +782,7 @@ fn processTest(boot_info: *const BootInfo) void { log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, usermode.write_count }); check("both init processes spawned on their own address spaces", spawned == 2); check("processes made repeated heartbeat syscalls (>=4)", usermode.write_count >= 4); - check("heartbeats came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23); + check("heartbeats came from user mode (CPL 3)", usermode.write_from_user); check("a kernel task coexisted with the processes (preemption)", proc_worker_ran); result(); } @@ -829,7 +829,7 @@ fn initTest(boot_info: *const BootInfo) void { const beat_ok = usermode.write_len >= prefix.len and eql(usermode.write_buf[0..prefix.len], prefix); check("init produced repeated heartbeats (>=2)", usermode.write_count >= 2); check("heartbeat text arrived intact", beat_ok); - check("heartbeats came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23); + check("heartbeats came from user mode (CPL 3)", usermode.write_from_user); check("init is still alive (did not exit)", usermode.exit_code == 0); result(); } diff --git a/src/kernel/usermode.zig b/src/kernel/usermode.zig index cdd4a3b..9d5e639 100644 --- a/src/kernel/usermode.zig +++ b/src/kernel/usermode.zig @@ -50,27 +50,27 @@ pub fn pfBlob() []const u8 { } /// What a ping syscall recorded — evidence for the test to assert on. -pub const Ping = struct { value: u64 = 0, cs: u64 = 0, ticks: u64 = 0 }; +pub const Ping = struct { value: u64 = 0, from_user: bool = false, ticks: u64 = 0 }; pub var pings = [_]Ping{.{}} ** 2; pub var ping_count: usize = 0; /// What write syscalls produced (accumulated), and the exit syscall's code. pub var write_buf: [256]u8 = undefined; pub var write_len: usize = 0; -pub var write_cs: u64 = 0; +pub var write_from_user: bool = false; pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests) pub var exit_code: u64 = 0; -/// The M3 syscall surface, dispatched on the saved user rax: +/// The M3 syscall surface, dispatched on the saved syscall number: /// 0 = exit(code) — end the caller (process: free its AS + reschedule; /// borrowed test thread: unwind to the kernel caller) -/// 1 = ping(value) — record rdi + the caller's CS + the current tick count +/// 1 = ping(value) — record the value + the caller's privilege + the tick count /// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT: /// 3 = sleep(ms) — block the caller for ms milliseconds /// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md) -/// replaces these later; rax is written back as the return value already, since -/// the entry paths restore user registers from the trap frame. One handler -/// serves both the int-0x80 gate and the syscall stub. +/// replaces these later; the result is written back into the trap frame already, +/// since the entry paths restore user registers from it. One handler serves both +/// syscall entry paths. /// /// Install it once at boot (before any user code runs) via `init`. pub fn init() void { @@ -78,23 +78,23 @@ pub fn init() void { } fn syscall(state: *arch.CpuState) void { - switch (state.rax) { + switch (arch.syscallNumber(state)) { 0 => { - exit_code = state.rdi; + exit_code = arch.syscallArg(state, 0); // A scheduled process frees its address space and reschedules; a // borrowed test thread unwinds back to the kernel that entered it. if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit(); }, 3 => { - sched.sleep(state.rdi); - state.rax = 0; + sched.sleep(arch.syscallArg(state, 0)); + arch.setSyscallResult(state, 0); }, 1 => { if (ping_count < pings.len) { - pings[ping_count] = .{ .value = state.rdi, .cs = state.cs, .ticks = arch.ticks() }; + pings[ping_count] = .{ .value = arch.syscallArg(state, 0), .from_user = arch.fromUser(state), .ticks = arch.ticks() }; ping_count += 1; } - state.rax = 0; + arch.setSyscallResult(state, 0); }, 2 => { // The pointer must lie inside the user region (image + stack) — @@ -105,22 +105,22 @@ fn syscall(state: *arch.CpuState) void { // hole in the region passes the check and the read #PFs -> on_fault // halts — a self-DoS, not an isolation break. Copy-in with fault // recovery is M3+ (with SMAP, once there's a reason to enable it). - const ptr = state.rdi; - const len = state.rsi; + const ptr = arch.syscallArg(state, 0); + const len = arch.syscallArg(state, 1); if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) { const src: [*]const u8 = @ptrFromInt(ptr); @memcpy(write_buf[0..len], src[0..len]); // keep the latest message write_len = len; - write_cs = state.cs; + write_from_user = arch.fromUser(state); write_count += 1; log.write("DANOS-INIT: "); log.write(src[0..len]); - state.rax = len; + arch.setSyscallResult(state, len); } else { - state.rax = @bitCast(@as(i64, -1)); + arch.setSyscallResult(state, @bitCast(@as(i64, -1))); } }, - else => state.rax = @bitCast(@as(i64, -1)), + else => arch.setSyscallResult(state, @bitCast(@as(i64, -1))), } } @@ -128,7 +128,7 @@ fn syscall(state: *arch.CpuState) void { fn resetRecords() void { ping_count = 0; write_len = 0; - write_cs = 0; + write_from_user = false; write_count = 0; exit_code = 0; } @@ -257,11 +257,11 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru return error.BadEntry; } -/// Load one page of a segment into address space `pml4`: a fresh frame, zeroed +/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed /// and filled through the physmap, mapped user-accessible with the segment's W^X. /// On a later failure the whole address space is torn down, which frees every /// frame mapped into it — so no per-page rollback list is needed here. -fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void { +fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void { const frame = pmm.alloc() orelse return error.OutOfMemory; const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame)); @memset(dst[0..page_size], 0); @@ -270,7 +270,7 @@ fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) Ini const n = @min(page_size, seg.filesz - page_off); @memcpy(dst[0..n], image[seg.off + page_off ..][0..n]); } - arch.mapUserPageInto(pml4, seg.vaddr + page_off, frame, seg.writable, seg.executable); + arch.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable); } /// Load a user ELF image into a fresh address space and spawn it as a scheduled @@ -286,15 +286,15 @@ pub fn spawnProcess(image: []const u8, priority: u3) InitError!void { const flags = sync.enter(); defer sync.leave(flags); - const pml4 = arch.createAddressSpace() orelse return error.OutOfMemory; - errdefer arch.destroyAddressSpace(pml4); + const aspace = arch.createAddressSpace() orelse return error.OutOfMemory; + errdefer arch.destroyAddressSpace(aspace); for (segs[0..parsed.count]) |seg| { - for (0..seg.pages()) |i| try loadPageInto(pml4, image, seg, i); + for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i); } const stack_frame = pmm.alloc() orelse return error.OutOfMemory; - arch.mapUserPageInto(pml4, stack_virt, stack_frame, true, false); // RW + NX + arch.mapUserPageInto(aspace, stack_virt, stack_frame, true, false); // RW + NX - if (!sched.spawnUserLocked(pml4, parsed.entry, stack_virt + page_size, priority)) + if (!sched.spawnUserLocked(aspace, parsed.entry, stack_virt + page_size, priority)) return error.OutOfMemory; } diff --git a/test/qemu_test.py b/test/qemu_test.py index 7d2918f..bb7228b 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -156,9 +156,9 @@ CASES = [ "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, # Isolation: a ring-3 read of a kernel-only page must #PF with error code - # 0x5 (present|user) at the user RIP. ([\s\S] spans lines; `.` doesn't.) + # 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.) {"name": "user-pf", - "expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*RIP\s*: 0x00007000000000", + "expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*IP\s*: 0x00007000000000", "fail": r"DANOS-TEST-RESULT: FAIL"}, # The real user binary: the bootloader ships sbin/init off the ESP, the # kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.