Rename the arch interface to arch-neutral terms
The generic kernel imports cpu.zig as @import("arch") but still spoke
x86: readCr3/readCr2, pml4 and rip/rsp parameters, lapicHz/tscHz,
ioapicEntry*, IST stacks, APIC ids. Rename the public interface so the
same names work for x86_64, aarch64, and riscv64:
- readCr3 -> activePageTable; pml4 params -> root; vectorName ->
exceptionName; postCode -> checkpoint; lapicHz/tscHz ->
timerClockHz/clockHz; ioapicEntry* -> irqRoute*; ist_stack_size/
setApIstStack -> fault_stack_size/setFaultStack; startSecondary
takes a hw_id (APIC id here; MPIDR/hart id elsewhere)
- New trap-frame accessors (instructionPointer, stackPointer,
fromUser, faultAddress, syscallNumber/syscallArg/setSyscallResult)
so the generic syscall dispatcher and fault printer never name an
x86 register; readCr2 folds into faultAddress (null unless #PF)
- Generic-kernel identifiers follow: Task.pml4 -> aspace, rsp -> sp,
user_rip/user_rsp -> user_ip/user_sp, PerCpu.apic_id -> hw_id
PlatformConfig fields and the ACPI apic_id stay as-is: they describe
hardware actually discovered on this platform, and another arch would
define its own. The user-mode tests now assert fromUser instead of
the exact CS selector; the user-pf expectation follows the fault
printer's RIP -> IP label. All 28 QEMU tests pass.
This commit is contained in:
+120
-59
@@ -20,6 +20,65 @@ const pcpu = @import("percpu.zig");
|
|||||||
/// The saved register/trap frame passed to a fault handler.
|
/// The saved register/trap frame passed to a fault handler.
|
||||||
pub const CpuState = idt.CpuState;
|
pub const CpuState = idt.CpuState;
|
||||||
|
|
||||||
|
// --- trap-frame accessors ---------------------------------------------------
|
||||||
|
// The frame's fields are x86_64 registers; the generic kernel reads it through
|
||||||
|
// these accessors so it never names one.
|
||||||
|
|
||||||
|
/// The interrupted/faulting instruction address (RIP here; ELR_EL1 on aarch64,
|
||||||
|
/// sepc on riscv64).
|
||||||
|
pub fn instructionPointer(state: *const CpuState) u64 {
|
||||||
|
return state.rip;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The interrupted stack pointer (RSP here).
|
||||||
|
pub fn stackPointer(state: *const CpuState) u64 {
|
||||||
|
return state.rsp;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the trap came from user mode (CPL 3 here; EL0 on aarch64, U-mode on
|
||||||
|
/// riscv64).
|
||||||
|
pub fn fromUser(state: *const CpuState) bool {
|
||||||
|
return state.cs & 3 == 3;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The faulting virtual address, if this trap is a page fault (CR2 here;
|
||||||
|
/// FAR_EL1 on aarch64, stval on riscv64). Null for any other exception.
|
||||||
|
pub fn faultAddress(state: *const CpuState) ?u64 {
|
||||||
|
if (state.vector != 14) return null;
|
||||||
|
return asm volatile ("mov %%cr2, %[out]"
|
||||||
|
: [out] "=r" (-> u64),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- syscall ABI ------------------------------------------------------------
|
||||||
|
// The System V-style register convention (number in rax, arguments in
|
||||||
|
// rdi/rsi/rdx/r10/r8/r9, result in rax), exposed positionally so the generic
|
||||||
|
// dispatcher never names a register.
|
||||||
|
|
||||||
|
/// The syscall number the user program passed.
|
||||||
|
pub fn syscallNumber(state: *const CpuState) u64 {
|
||||||
|
return state.rax;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Positional syscall argument `n`.
|
||||||
|
pub fn syscallArg(state: *const CpuState, n: u8) u64 {
|
||||||
|
return switch (n) {
|
||||||
|
0 => state.rdi,
|
||||||
|
1 => state.rsi,
|
||||||
|
2 => state.rdx,
|
||||||
|
3 => state.r10,
|
||||||
|
4 => state.r8,
|
||||||
|
5 => state.r9,
|
||||||
|
else => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the syscall's return value into the frame — the entry paths restore
|
||||||
|
/// user registers from it.
|
||||||
|
pub fn setSyscallResult(state: *CpuState, value: u64) void {
|
||||||
|
state.rax = value;
|
||||||
|
}
|
||||||
|
|
||||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||||
/// so it can be the very first thing called.
|
/// so it can be the very first thing called.
|
||||||
pub fn serialInit() void {
|
pub fn serialInit() void {
|
||||||
@@ -31,10 +90,11 @@ pub fn serialWrite(bytes: []const u8) void {
|
|||||||
serial.write(bytes);
|
serial.write(bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Emit a one-byte checkpoint to the POST diagnostic port (0x80). A POST card or
|
/// Emit a one-byte progress checkpoint to whatever hardware debug sink the
|
||||||
/// BMC displays it; it's the last-resort progress signal when there's no text
|
/// platform has — here the POST diagnostic port (0x80), which a POST card or BMC
|
||||||
/// output at all. Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
/// displays. The last-resort progress signal when there's no text output at all.
|
||||||
pub fn postCode(code: u8) void {
|
/// Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
||||||
|
pub fn checkpoint(code: u8) void {
|
||||||
io.outb(0x80, code);
|
io.outb(0x80, code);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -68,21 +128,22 @@ pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) vo
|
|||||||
paging.init(allocFrame, freeFrame, boot_info);
|
paging.init(allocFrame, freeFrame, boot_info);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a new address space (returns its physical PML4, or null). Shares the
|
/// Create a new address space (returns the physical address of its root table —
|
||||||
/// kernel's higher half; the user (low) half starts empty.
|
/// the PML4 here — or null). Shares the kernel's higher half; the user (low)
|
||||||
|
/// half starts empty.
|
||||||
pub fn createAddressSpace() ?u64 {
|
pub fn createAddressSpace() ?u64 {
|
||||||
return paging.createAddressSpace();
|
return paging.createAddressSpace();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Free an address space and everything mapped in its user half. Caller must not
|
/// Free an address space and everything mapped in its user half. Caller must not
|
||||||
/// be running on it.
|
/// be running on it.
|
||||||
pub fn destroyAddressSpace(pml4: u64) void {
|
pub fn destroyAddressSpace(root: u64) void {
|
||||||
paging.destroyAddressSpace(pml4);
|
paging.destroyAddressSpace(root);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a ring-3 page into address space `pml4` (W^X is the caller's contract).
|
/// Map a user page into address space `root` (W^X is the caller's contract).
|
||||||
pub fn mapUserPageInto(pml4: u64, virt: u64, phys: u64, writable: bool, executable: bool) void {
|
pub fn mapUserPageInto(root: u64, virt: u64, phys: u64, writable: bool, executable: bool) void {
|
||||||
paging.mapUserInto(pml4, virt, phys, writable, executable);
|
paging.mapUserInto(root, virt, phys, writable, executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||||
@@ -102,9 +163,17 @@ pub fn kernelPageTable() u64 {
|
|||||||
return paging.kernelPml4();
|
return paging.kernelPml4();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Switch the active address space (load CR3 with a physical PML4).
|
/// Switch the active address space (load CR3 with a physical root table).
|
||||||
pub fn loadPageTable(pml4: u64) void {
|
pub fn loadPageTable(root: u64) void {
|
||||||
paging.loadCr3(pml4);
|
paging.loadCr3(root);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The physical root of the currently active page tables (CR3 here; TTBR0/satp
|
||||||
|
/// elsewhere).
|
||||||
|
pub fn activePageTable() u64 {
|
||||||
|
return asm volatile ("mov %%cr3, %[out]"
|
||||||
|
: [out] "=r" (-> u64),
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
||||||
@@ -140,12 +209,12 @@ extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c)
|
|||||||
/// `enter_user` had returned (defined in isr.s). Called by the exit syscall.
|
/// `enter_user` had returned (defined in isr.s). Called by the exit syscall.
|
||||||
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
||||||
|
|
||||||
/// Run user code at `rip` with stack `rsp` on this core (`cpu` = the caller's CPU
|
/// Run user code at `entry` with stack `stack_top` on this core (`cpu` = the
|
||||||
/// index; the arch layer can't ask the scheduler). Returns after the user program
|
/// caller's CPU index; the arch layer can't ask the scheduler). Returns after the
|
||||||
/// exits via syscall. Interrupts are disabled on return (the exit arrives through
|
/// user program exits via syscall. Interrupts are disabled on return (the exit
|
||||||
/// an interrupt gate) — the caller re-enables.
|
/// arrives through an interrupt gate) — the caller re-enables.
|
||||||
pub fn enterUser(cpu: usize, rip: u64, rsp: u64) void {
|
pub fn enterUser(cpu: usize, entry: u64, stack_top: u64) void {
|
||||||
enter_user(rip, rsp, tss.rsp0Ptr(cpu));
|
enter_user(entry, stack_top, tss.rsp0Ptr(cpu));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Never returns to the user program: unwind to the kernel context that called
|
/// Never returns to the user program: unwind to the kernel context that called
|
||||||
@@ -154,19 +223,12 @@ pub fn userExit() noreturn {
|
|||||||
user_exit_to_kernel();
|
user_exit_to_kernel();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Register the handler for the ring-3 syscall gate (int 0x80, vector 128). The
|
/// Register the handler for the user syscall gate (int 0x80, vector 128). The
|
||||||
/// handler may write the trap frame (e.g. rax as the return value).
|
/// handler may write the trap frame (see `setSyscallResult`).
|
||||||
pub fn setSyscallHandler(handler: *const fn (*idt.CpuState) void) void {
|
pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void {
|
||||||
idt.setSyscallHandler(handler);
|
idt.setSyscallHandler(handler);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// CR3 holds the physical address of the active top-level page table.
|
|
||||||
pub fn readCr3() u64 {
|
|
||||||
return asm volatile ("mov %%cr3, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
||||||
/// core calls this once, after its GDT is in place (a GS *selector* reload would
|
/// core calls this once, after its GDT is in place (a GS *selector* reload would
|
||||||
/// clobber the base). See percpu.zig for the swapgs discipline.
|
/// clobber the base). See percpu.zig for the swapgs discipline.
|
||||||
@@ -190,15 +252,15 @@ pub fn setTrampolinePage(phys: u64) void {
|
|||||||
smp.setTrampolinePage(phys);
|
smp.setTrampolinePage(phys);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it
|
/// Wake the core with hardware id `hw_id` (its Local APIC id here; MPIDR on
|
||||||
/// `stack_top` and its per-CPU pointer `percpu`; it adopts the current (kernel) page
|
/// aarch64, hart id on riscv64) as dense CPU `index`, giving it `stack_top` and
|
||||||
/// tables. Returns false if it doesn't come online within the timeout. Blocks until
|
/// its per-CPU pointer `percpu`; it adopts the kernel page tables. Returns false
|
||||||
/// the core reports in.
|
/// if it doesn't come online within the timeout. Blocks until the core reports in.
|
||||||
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
|
pub fn startSecondary(hw_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
|
||||||
// The AP adopts the kernel page tables explicitly — never the caller's live
|
// The AP adopts the kernel page tables explicitly — never the caller's live
|
||||||
// CR3, which a future re-wake from a core running a process would make a
|
// CR3, which a future re-wake from a core running a process would make a
|
||||||
// process address space.
|
// process address space.
|
||||||
return smp.startAp(apic_id, stack_top, percpu, index, paging.kernelPml4());
|
return smp.startAp(hw_id, stack_top, percpu, index, paging.kernelPml4());
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Register the generic entry a woken AP jumps to once its arch state is up (its own
|
/// Register the generic entry a woken AP jumps to once its arch state is up (its own
|
||||||
@@ -207,11 +269,12 @@ pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
|||||||
smp.setSecondaryEntry(entry);
|
smp.setSecondaryEntry(entry);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bytes the kernel should allocate for an AP's IST (double-fault) stack, and where
|
/// Bytes the kernel should allocate for a secondary core's dedicated fault stack
|
||||||
/// to record its top before waking the core. The stack is heap-allocated per online
|
/// (the IST double-fault stack here), and where to record its top before waking
|
||||||
/// AP (the BSP's is static — it's needed before the allocator exists). See tss.zig.
|
/// the core. The stack is heap-allocated per online core (the boot CPU's is
|
||||||
pub const ist_stack_size = tss.ist_stack_size;
|
/// static — it's needed before the allocator exists). See tss.zig.
|
||||||
pub fn setApIstStack(cpu: usize, top: usize) void {
|
pub const fault_stack_size = tss.ist_stack_size;
|
||||||
|
pub fn setFaultStack(cpu: usize, top: usize) void {
|
||||||
tss.setApIstStack(cpu, top);
|
tss.setApIstStack(cpu, top);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -274,11 +337,12 @@ pub fn timerCalibrationSource() []const u8 {
|
|||||||
return apic.calibrationSource();
|
return apic.calibrationSource();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// I/O APIC diagnostics (for boot logging / verification).
|
/// External-interrupt-router diagnostics, for boot logging / verification (the
|
||||||
pub fn ioapicEntryCount() u32 {
|
/// I/O APIC's redirection entries here; a GIC distributor or PLIC elsewhere).
|
||||||
|
pub fn irqRouteCount() u32 {
|
||||||
return ioapic.entryCount();
|
return ioapic.entryCount();
|
||||||
}
|
}
|
||||||
pub fn ioapicEntryLow(n: u32) u32 {
|
pub fn irqRouteRaw(n: u32) u32 {
|
||||||
return ioapic.entryLow(n);
|
return ioapic.entryLow(n);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -309,11 +373,15 @@ pub fn millis() u64 {
|
|||||||
return apic.millis();
|
return apic.millis();
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
/// Measured frequency of the tick timer's input clock (the LAPIC timer here), in
|
||||||
pub fn lapicHz() u64 {
|
/// Hz, from calibration.
|
||||||
|
pub fn timerClockHz() u64 {
|
||||||
return apic.lapicHz();
|
return apic.lapicHz();
|
||||||
}
|
}
|
||||||
pub fn tscHz() u64 {
|
|
||||||
|
/// Measured frequency of the monotonic clock's underlying counter (the TSC here;
|
||||||
|
/// CNTVCT on aarch64, `time` on riscv64), in Hz.
|
||||||
|
pub fn clockHz() u64 {
|
||||||
return apic.tscHz();
|
return apic.tscHz();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -359,8 +427,8 @@ pub fn setTickHook(hook: *const fn () void) void {
|
|||||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||||
|
|
||||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
pub fn switchContext(old_sp: *usize, new_sp: usize) void {
|
||||||
switch_context(old_rsp, new_rsp);
|
switch_context(old_sp, new_sp);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the initial stack for a new task so that switching to it lands in
|
/// Build the initial stack for a new task so that switching to it lands in
|
||||||
@@ -393,8 +461,8 @@ pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
|||||||
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
|
/// ring 0; the pushed RFLAGS re-enables them in ring 3.
|
||||||
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
|
extern fn jump_to_user(rip: u64, rsp: u64) callconv(.c) noreturn;
|
||||||
|
|
||||||
pub fn jumpToUser(rip: u64, rsp: u64) noreturn {
|
pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
|
||||||
jump_to_user(rip, rsp);
|
jump_to_user(entry, stack_top);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||||
@@ -404,7 +472,7 @@ pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// A human-readable name for a CPU exception vector.
|
/// A human-readable name for a CPU exception vector.
|
||||||
pub fn vectorName(vector: u64) []const u8 {
|
pub fn exceptionName(vector: u64) []const u8 {
|
||||||
return idt.vectorName(vector);
|
return idt.vectorName(vector);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -430,13 +498,6 @@ pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
|
||||||
pub fn readCr2() u64 {
|
|
||||||
return asm volatile ("mov %%cr2, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
||||||
/// docs/halting.md for the full reasoning.
|
/// docs/halting.md for the full reasoning.
|
||||||
|
|||||||
+1
-1
@@ -52,7 +52,7 @@ pub fn print(comptime fmt: []const u8, args: anytype) void {
|
|||||||
/// progress channel for when there is no text output at all. Independent of the
|
/// progress channel for when there is no text output at all. Independent of the
|
||||||
/// sink list, so it works even before any sink is registered.
|
/// sink list, so it works even before any sink is registered.
|
||||||
pub fn checkpoint(code: u8) void {
|
pub fn checkpoint(code: u8) void {
|
||||||
arch.postCode(code);
|
arch.checkpoint(code);
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- persistent panic breadcrumb -------------------------------------------
|
// --- persistent panic breadcrumb -------------------------------------------
|
||||||
|
|||||||
+12
-12
@@ -131,7 +131,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
arch.enablePaging(pmm.alloc, pmm.free, boot_info);
|
arch.enablePaging(pmm.alloc, pmm.free, boot_info);
|
||||||
log.checkpoint(cp_paging);
|
log.checkpoint(cp_paging);
|
||||||
log.print("\ndanos: paging enabled\n", .{});
|
log.print("\ndanos: paging enabled\n", .{});
|
||||||
log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
log.print(" page tables: root = 0x{x:0>16}\n", .{arch.activePageTable()});
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||||
@@ -216,7 +216,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
} else {
|
} else {
|
||||||
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
||||||
}
|
}
|
||||||
log.print(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) });
|
log.print(" ioapic : base 0x{x}, {d} inputs (masked); route0 raw 0x{x}\n", .{ ioapic_base, arch.irqRouteCount(), arch.irqRouteRaw(0) });
|
||||||
const cores = platform.cpus();
|
const cores = platform.cpus();
|
||||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||||
if (platform.cpusDropped() > 0)
|
if (platform.cpusDropped() > 0)
|
||||||
@@ -240,7 +240,7 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
|||||||
arch.startTimer();
|
arch.startTimer();
|
||||||
arch.enableInterrupts();
|
arch.enableInterrupts();
|
||||||
log.checkpoint(cp_timer);
|
log.checkpoint(cp_timer);
|
||||||
log.print("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() });
|
log.print("danos: timer online ({d} Hz tick; timer clock {d} MHz, clock {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.timerClockHz() / 1_000_000, arch.clockHz() / 1_000_000, arch.timerCalibrationSource() });
|
||||||
|
|
||||||
// Wake the other cores (application processors). A no-op on a single-core
|
// Wake the other cores (application processors). A no-op on a single-core
|
||||||
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
// machine; on SMP each AP climbs to long mode and reports in (docs/smp.md).
|
||||||
@@ -309,13 +309,13 @@ fn bringUpSecondaries() void {
|
|||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
||||||
// This core's IST (double-fault) stack — allocated only now that the core is
|
// This core's dedicated fault stack — allocated only now that the core is
|
||||||
// real, rather than reserved statically for every possible core.
|
// real, rather than reserved statically for every possible core.
|
||||||
const ist = heap.allocator().alloc(u8, arch.ist_stack_size) catch {
|
const fault_stack = heap.allocator().alloc(u8, arch.fault_stack_size) catch {
|
||||||
log.print(" cpu apic_id {d}: no IST stack; skipped\n", .{core.apic_id});
|
log.print(" cpu apic_id {d}: no fault stack; skipped\n", .{core.apic_id});
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
arch.setApIstStack(index, (@intFromPtr(ist.ptr) + ist.len) & ~@as(usize, 15));
|
arch.setFaultStack(index, (@intFromPtr(fault_stack.ptr) + fault_stack.len) & ~@as(usize, 15));
|
||||||
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
||||||
var attempt: u32 = 1;
|
var attempt: u32 = 1;
|
||||||
while (attempt <= max_wake_attempts) : (attempt += 1) {
|
while (attempt <= max_wake_attempts) : (attempt += 1) {
|
||||||
@@ -364,14 +364,14 @@ fn onException(state: *const arch.CpuState) noreturn {
|
|||||||
const core = scheduler.currentCpuIndex();
|
const core = scheduler.currentCpuIndex();
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
||||||
// top of the diagnostic log.
|
// top of the diagnostic log.
|
||||||
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.vectorName(state.vector), state.vector });
|
statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.exceptionName(state.vector), state.vector });
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||||
statusPrint(" RIP : 0x{x:0>16}\n", .{state.rip});
|
statusPrint(" IP : 0x{x:0>16}\n", .{arch.instructionPointer(state)});
|
||||||
statusPrint(" RSP : 0x{x:0>16}\n", .{state.rsp});
|
statusPrint(" SP : 0x{x:0>16}\n", .{arch.stackPointer(state)});
|
||||||
if (state.vector == 14) statusPrint(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
|
if (arch.faultAddress(state)) |addr| statusPrint(" fault addr : 0x{x:0>16}\n", .{addr});
|
||||||
|
|
||||||
var buf: [128]u8 = undefined;
|
var buf: [128]u8 = undefined;
|
||||||
log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, core, state.rip }) catch "cpu exception");
|
log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ arch.exceptionName(state.vector), state.vector, core, arch.instructionPointer(state) }) catch "cpu exception");
|
||||||
arch.halt();
|
arch.halt();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+44
-44
@@ -36,16 +36,16 @@ const Task = struct {
|
|||||||
id: u32 = 0,
|
id: u32 = 0,
|
||||||
state: State = .free,
|
state: State = .free,
|
||||||
priority: Priority = 0,
|
priority: Priority = 0,
|
||||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
sp: usize = 0, // saved stack pointer, valid while not running
|
||||||
stack: []u8 = &.{},
|
stack: []u8 = &.{},
|
||||||
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
|
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
|
||||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||||
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
||||||
// Physical PML4 of this task's address space, or 0 for a kernel task (which
|
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||||
// runs on the shared kernel page tables). A user task carries its own.
|
// runs on the shared kernel page tables). A user task carries its own.
|
||||||
pml4: u64 = 0,
|
aspace: u64 = 0,
|
||||||
user_rip: u64 = 0, // ring-3 entry point (user task only)
|
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||||
user_rsp: u64 = 0, // ring-3 stack pointer (user task only)
|
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||||
next: ?*Task = null, // ready-queue link
|
next: ?*Task = null, // ready-queue link
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -66,10 +66,10 @@ var next_id: u32 = 1;
|
|||||||
pub const PerCpu = struct {
|
pub const PerCpu = struct {
|
||||||
current: *Task = undefined, // the task running on this core
|
current: *Task = undefined, // the task running on this core
|
||||||
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
||||||
apic_id: u32 = 0, // the core's Local APIC id
|
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
||||||
index: u32 = 0, // dense 0-based core index
|
index: u32 = 0, // dense 0-based core index
|
||||||
online: bool = false, // has this core finished bring-up?
|
online: bool = false, // has this core finished bring-up?
|
||||||
loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core
|
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
||||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||||
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
|
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
|
||||||
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
|
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
|
||||||
@@ -105,7 +105,7 @@ var preemption_enabled = true;
|
|||||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||||
pub fn init(boot_priority: Priority) void {
|
pub fn init(boot_priority: Priority) void {
|
||||||
const pc = &cpus[0];
|
const pc = &cpus[0];
|
||||||
pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() };
|
pc.* = .{ .index = 0, .online = true, .loaded_aspace = arch.kernelPageTable() };
|
||||||
arch.setCpuLocal(0, @intFromPtr(pc));
|
arch.setCpuLocal(0, @intFromPtr(pc));
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||||
pc.current = &tasks[0];
|
pc.current = &tasks[0];
|
||||||
@@ -120,12 +120,12 @@ fn idle() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
||||||
/// `index` (1-based; 0 is the BSP) with Local APIC id `apic_id`, and return a
|
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
|
||||||
/// pointer the arch bring-up hands to the core (it publishes it in its GS base).
|
/// pointer the arch bring-up hands to the core (it publishes it in its GS base).
|
||||||
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
||||||
pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu {
|
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
||||||
const pc = &cpus[index];
|
const pc = &cpus[index];
|
||||||
pc.* = .{ .index = @intCast(index), .apic_id = apic_id, .online = false };
|
pc.* = .{ .index = @intCast(index), .hw_id = hw_id, .online = false };
|
||||||
return pc;
|
return pc;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -144,7 +144,7 @@ pub fn secondaryMain() callconv(.c) noreturn {
|
|||||||
pc.current = t;
|
pc.current = t;
|
||||||
pc.idle = t;
|
pc.idle = t;
|
||||||
pc.online = true;
|
pc.online = true;
|
||||||
pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
pc.loaded_aspace = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
|
|
||||||
arch.enableInterrupts(); // the timer now preempts this idle context into work
|
arch.enableInterrupts(); // the timer now preempts this idle context into work
|
||||||
@@ -230,13 +230,13 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
|||||||
return ok;
|
return ok;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn a **user** task: a task with its own address space (`pml4`) that starts
|
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||||
/// in ring 3 at `entry_rip` on `user_rsp`. It gets a fresh kernel stack for
|
/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for
|
||||||
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
||||||
/// Returns false (creating nothing) if the table is full or out of memory.
|
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||||
/// **Caller must hold the kernel lock** (the loader that builds `pml4` holds it
|
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||||
/// across the whole spawn, so the address space and the task appear atomically).
|
/// across the whole spawn, so the address space and the task appear atomically).
|
||||||
pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Priority) bool {
|
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool {
|
||||||
const t = freeSlot() orelse return false;
|
const t = freeSlot() orelse return false;
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||||
t.* = .{
|
t.* = .{
|
||||||
@@ -244,16 +244,16 @@ pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Prior
|
|||||||
.state = .ready,
|
.state = .ready,
|
||||||
.priority = priority,
|
.priority = priority,
|
||||||
.stack = stack,
|
.stack = stack,
|
||||||
.pml4 = pml4,
|
.aspace = aspace,
|
||||||
.user_rip = entry_rip,
|
.user_ip = entry,
|
||||||
.user_rsp = user_rsp,
|
.user_sp = user_sp,
|
||||||
};
|
};
|
||||||
next_id += 1;
|
next_id += 1;
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
t.kstack_top = top;
|
t.kstack_top = top;
|
||||||
// First switch-in lands in startUserTask (no register smuggling — it reads
|
// First switch-in lands in startUserTask (no register smuggling — it reads
|
||||||
// the ring-3 entry/stack from the Task itself).
|
// the user entry/stack from the Task itself).
|
||||||
t.rsp = arch.initTaskStack(top, @intFromPtr(&startUserTask));
|
t.sp = arch.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||||
enqueue(t);
|
enqueue(t);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -265,8 +265,8 @@ pub fn spawnUserLocked(pml4: u64, entry_rip: u64, user_rsp: u64, priority: Prior
|
|||||||
fn startUserTask() void {
|
fn startUserTask() void {
|
||||||
const t = cur();
|
const t = cur();
|
||||||
var buf: [96]u8 = undefined;
|
var buf: [96]u8 = undefined;
|
||||||
arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask rip=0x{x} rsp=0x{x} pml4=0x{x} kstack=0x{x}\n", .{ t.user_rip, t.user_rsp, t.pml4, t.kstack_top }) catch "");
|
arch.serialWrite(std.fmt.bufPrint(&buf, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
||||||
arch.jumpToUser(t.user_rip, t.user_rsp); // noreturn
|
arch.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||||
@@ -279,7 +279,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
|||||||
next_id += 1;
|
next_id += 1;
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
t.kstack_top = top;
|
t.kstack_top = top;
|
||||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
t.sp = arch.initTaskStack(top, @intFromPtr(entry));
|
||||||
enqueue(t);
|
enqueue(t);
|
||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
@@ -309,26 +309,26 @@ fn schedule() void {
|
|||||||
};
|
};
|
||||||
next.state = .running;
|
next.state = .running;
|
||||||
pc.current = next;
|
pc.current = next;
|
||||||
if (next != prev) switchTo(pc, &prev.rsp, next);
|
if (next != prev) switchTo(pc, &prev.sp, next);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||||
/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when
|
/// user-mode interrupt lands on a good stack) and its address space (only when
|
||||||
/// it differs from what's loaded — every CR3 write is a full TLB flush), then
|
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
||||||
/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from
|
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
||||||
/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so
|
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
||||||
/// this is a no-op beyond the register switch for a pure-kernel workload. The
|
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
||||||
/// big kernel lock is held and interrupts are off throughout, so no interrupt
|
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
||||||
/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing
|
/// interrupt can observe a half-updated (kernel stack, address space) pair.
|
||||||
/// task's stack pointer.
|
/// `save_sp` receives the outgoing task's stack pointer.
|
||||||
fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void {
|
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||||
if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top);
|
if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top);
|
||||||
const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable();
|
const want = if (next.aspace != 0) next.aspace else arch.kernelPageTable();
|
||||||
if (want != pc.loaded_pml4) {
|
if (want != pc.loaded_aspace) {
|
||||||
arch.loadPageTable(want);
|
arch.loadPageTable(want);
|
||||||
pc.loaded_pml4 = want;
|
pc.loaded_aspace = want;
|
||||||
}
|
}
|
||||||
arch.switchContext(save_rsp, next.rsp);
|
arch.switchContext(save_sp, next.sp);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Voluntarily give up the CPU to the next ready task.
|
/// Voluntarily give up the CPU to the next ready task.
|
||||||
@@ -478,15 +478,15 @@ pub fn exitUser() noreturn {
|
|||||||
_ = sync.enter();
|
_ = sync.enter();
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
const dying = pc.current;
|
const dying = pc.current;
|
||||||
const as = dying.pml4;
|
const as = dying.aspace;
|
||||||
if (as != 0) {
|
if (as != 0) {
|
||||||
const kpml4 = arch.kernelPageTable();
|
const kroot = arch.kernelPageTable();
|
||||||
arch.loadPageTable(kpml4); // off the process tables before freeing them
|
arch.loadPageTable(kroot); // off the process tables before freeing them
|
||||||
pc.loaded_pml4 = kpml4;
|
pc.loaded_aspace = kroot;
|
||||||
arch.destroyAddressSpace(as);
|
arch.destroyAddressSpace(as);
|
||||||
}
|
}
|
||||||
dying.state = .free;
|
dying.state = .free;
|
||||||
dying.pml4 = 0;
|
dying.aspace = 0;
|
||||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||||
next.state = .running;
|
next.state = .running;
|
||||||
pc.current = next;
|
pc.current = next;
|
||||||
@@ -497,7 +497,7 @@ pub fn exitUser() noreturn {
|
|||||||
|
|
||||||
/// Whether the running task is a user process (has its own address space).
|
/// Whether the running task is a user process (has its own address space).
|
||||||
pub fn currentIsUserProcess() bool {
|
pub fn currentIsUserProcess() bool {
|
||||||
return cur().pml4 != 0;
|
return cur().aspace != 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
pub fn currentId() u32 {
|
||||||
|
|||||||
+11
-11
@@ -166,9 +166,9 @@ fn smoke(boot_info: *const BootInfo) void {
|
|||||||
if (b) |p| pmm.free(p);
|
if (b) |p| pmm.free(p);
|
||||||
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
||||||
|
|
||||||
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
// Paging is active on our own tables (the root is non-zero and page-aligned).
|
||||||
const cr3 = arch.readCr3();
|
const root = arch.activePageTable();
|
||||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
check("paging active (page-table root set)", root != 0 and root % danos.page_size == 0);
|
||||||
|
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -324,10 +324,10 @@ fn heapTest() void {
|
|||||||
fn clock() void {
|
fn clock() void {
|
||||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
log("DANOS-TEST-BEGIN: clock\n", .{});
|
||||||
|
|
||||||
const lapic = arch.lapicHz();
|
const timer_clock = arch.timerClockHz();
|
||||||
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
check("timer clock frequency measured", timer_clock > 1_000_000 and timer_clock < 100_000_000_000);
|
||||||
const tsc = arch.tscHz();
|
const clock_hz = arch.clockHz();
|
||||||
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
check("monotonic clock frequency measured", clock_hz > 100_000_000 and clock_hz < 100_000_000_000);
|
||||||
|
|
||||||
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
||||||
const start_ticks = arch.ticks();
|
const start_ticks = arch.ticks();
|
||||||
@@ -729,7 +729,7 @@ fn userTest() void {
|
|||||||
check("user program ran and exited (ring-3 round trip)", ran);
|
check("user program ran and exited (ring-3 round trip)", ran);
|
||||||
check("two ping syscalls received", usermode.ping_count == 2);
|
check("two ping syscalls received", usermode.ping_count == 2);
|
||||||
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
|
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
|
||||||
check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23);
|
check("syscalls came from user mode (CPL 3)", usermode.pings[0].from_user and usermode.pings[1].from_user);
|
||||||
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
|
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -762,7 +762,7 @@ fn processTest(boot_info: *const BootInfo) void {
|
|||||||
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
||||||
|
|
||||||
usermode.write_count = 0;
|
usermode.write_count = 0;
|
||||||
usermode.write_cs = 0;
|
usermode.write_from_user = false;
|
||||||
proc_worker_run = true;
|
proc_worker_run = true;
|
||||||
proc_worker_ran = false;
|
proc_worker_ran = false;
|
||||||
sched.spawn(procWorker, 4); // kernel task at the processes' priority
|
sched.spawn(procWorker, 4); // kernel task at the processes' priority
|
||||||
@@ -782,7 +782,7 @@ fn processTest(boot_info: *const BootInfo) void {
|
|||||||
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, usermode.write_count });
|
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, usermode.write_count });
|
||||||
check("both init processes spawned on their own address spaces", spawned == 2);
|
check("both init processes spawned on their own address spaces", spawned == 2);
|
||||||
check("processes made repeated heartbeat syscalls (>=4)", usermode.write_count >= 4);
|
check("processes made repeated heartbeat syscalls (>=4)", usermode.write_count >= 4);
|
||||||
check("heartbeats came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23);
|
check("heartbeats came from user mode (CPL 3)", usermode.write_from_user);
|
||||||
check("a kernel task coexisted with the processes (preemption)", proc_worker_ran);
|
check("a kernel task coexisted with the processes (preemption)", proc_worker_ran);
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
@@ -829,7 +829,7 @@ fn initTest(boot_info: *const BootInfo) void {
|
|||||||
const beat_ok = usermode.write_len >= prefix.len and eql(usermode.write_buf[0..prefix.len], prefix);
|
const beat_ok = usermode.write_len >= prefix.len and eql(usermode.write_buf[0..prefix.len], prefix);
|
||||||
check("init produced repeated heartbeats (>=2)", usermode.write_count >= 2);
|
check("init produced repeated heartbeats (>=2)", usermode.write_count >= 2);
|
||||||
check("heartbeat text arrived intact", beat_ok);
|
check("heartbeat text arrived intact", beat_ok);
|
||||||
check("heartbeats came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23);
|
check("heartbeats came from user mode (CPL 3)", usermode.write_from_user);
|
||||||
check("init is still alive (did not exit)", usermode.exit_code == 0);
|
check("init is still alive (did not exit)", usermode.exit_code == 0);
|
||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|||||||
+28
-28
@@ -50,27 +50,27 @@ pub fn pfBlob() []const u8 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// What a ping syscall recorded — evidence for the test to assert on.
|
/// What a ping syscall recorded — evidence for the test to assert on.
|
||||||
pub const Ping = struct { value: u64 = 0, cs: u64 = 0, ticks: u64 = 0 };
|
pub const Ping = struct { value: u64 = 0, from_user: bool = false, ticks: u64 = 0 };
|
||||||
pub var pings = [_]Ping{.{}} ** 2;
|
pub var pings = [_]Ping{.{}} ** 2;
|
||||||
pub var ping_count: usize = 0;
|
pub var ping_count: usize = 0;
|
||||||
|
|
||||||
/// What write syscalls produced (accumulated), and the exit syscall's code.
|
/// What write syscalls produced (accumulated), and the exit syscall's code.
|
||||||
pub var write_buf: [256]u8 = undefined;
|
pub var write_buf: [256]u8 = undefined;
|
||||||
pub var write_len: usize = 0;
|
pub var write_len: usize = 0;
|
||||||
pub var write_cs: u64 = 0;
|
pub var write_from_user: bool = false;
|
||||||
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
||||||
pub var exit_code: u64 = 0;
|
pub var exit_code: u64 = 0;
|
||||||
|
|
||||||
/// The M3 syscall surface, dispatched on the saved user rax:
|
/// The M3 syscall surface, dispatched on the saved syscall number:
|
||||||
/// 0 = exit(code) — end the caller (process: free its AS + reschedule;
|
/// 0 = exit(code) — end the caller (process: free its AS + reschedule;
|
||||||
/// borrowed test thread: unwind to the kernel caller)
|
/// borrowed test thread: unwind to the kernel caller)
|
||||||
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
|
/// 1 = ping(value) — record the value + the caller's privilege + the tick count
|
||||||
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
|
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
|
||||||
/// 3 = sleep(ms) — block the caller for ms milliseconds
|
/// 3 = sleep(ms) — block the caller for ms milliseconds
|
||||||
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
|
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
|
||||||
/// replaces these later; rax is written back as the return value already, since
|
/// replaces these later; the result is written back into the trap frame already,
|
||||||
/// the entry paths restore user registers from the trap frame. One handler
|
/// since the entry paths restore user registers from it. One handler serves both
|
||||||
/// serves both the int-0x80 gate and the syscall stub.
|
/// syscall entry paths.
|
||||||
///
|
///
|
||||||
/// Install it once at boot (before any user code runs) via `init`.
|
/// Install it once at boot (before any user code runs) via `init`.
|
||||||
pub fn init() void {
|
pub fn init() void {
|
||||||
@@ -78,23 +78,23 @@ pub fn init() void {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn syscall(state: *arch.CpuState) void {
|
fn syscall(state: *arch.CpuState) void {
|
||||||
switch (state.rax) {
|
switch (arch.syscallNumber(state)) {
|
||||||
0 => {
|
0 => {
|
||||||
exit_code = state.rdi;
|
exit_code = arch.syscallArg(state, 0);
|
||||||
// A scheduled process frees its address space and reschedules; a
|
// A scheduled process frees its address space and reschedules; a
|
||||||
// borrowed test thread unwinds back to the kernel that entered it.
|
// borrowed test thread unwinds back to the kernel that entered it.
|
||||||
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
|
if (sched.currentIsUserProcess()) sched.exitUser() else arch.userExit();
|
||||||
},
|
},
|
||||||
3 => {
|
3 => {
|
||||||
sched.sleep(state.rdi);
|
sched.sleep(arch.syscallArg(state, 0));
|
||||||
state.rax = 0;
|
arch.setSyscallResult(state, 0);
|
||||||
},
|
},
|
||||||
1 => {
|
1 => {
|
||||||
if (ping_count < pings.len) {
|
if (ping_count < pings.len) {
|
||||||
pings[ping_count] = .{ .value = state.rdi, .cs = state.cs, .ticks = arch.ticks() };
|
pings[ping_count] = .{ .value = arch.syscallArg(state, 0), .from_user = arch.fromUser(state), .ticks = arch.ticks() };
|
||||||
ping_count += 1;
|
ping_count += 1;
|
||||||
}
|
}
|
||||||
state.rax = 0;
|
arch.setSyscallResult(state, 0);
|
||||||
},
|
},
|
||||||
2 => {
|
2 => {
|
||||||
// The pointer must lie inside the user region (image + stack) —
|
// The pointer must lie inside the user region (image + stack) —
|
||||||
@@ -105,22 +105,22 @@ fn syscall(state: *arch.CpuState) void {
|
|||||||
// hole in the region passes the check and the read #PFs -> on_fault
|
// hole in the region passes the check and the read #PFs -> on_fault
|
||||||
// halts — a self-DoS, not an isolation break. Copy-in with fault
|
// halts — a self-DoS, not an isolation break. Copy-in with fault
|
||||||
// recovery is M3+ (with SMAP, once there's a reason to enable it).
|
// recovery is M3+ (with SMAP, once there's a reason to enable it).
|
||||||
const ptr = state.rdi;
|
const ptr = arch.syscallArg(state, 0);
|
||||||
const len = state.rsi;
|
const len = arch.syscallArg(state, 1);
|
||||||
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
|
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
|
||||||
const src: [*]const u8 = @ptrFromInt(ptr);
|
const src: [*]const u8 = @ptrFromInt(ptr);
|
||||||
@memcpy(write_buf[0..len], src[0..len]); // keep the latest message
|
@memcpy(write_buf[0..len], src[0..len]); // keep the latest message
|
||||||
write_len = len;
|
write_len = len;
|
||||||
write_cs = state.cs;
|
write_from_user = arch.fromUser(state);
|
||||||
write_count += 1;
|
write_count += 1;
|
||||||
log.write("DANOS-INIT: ");
|
log.write("DANOS-INIT: ");
|
||||||
log.write(src[0..len]);
|
log.write(src[0..len]);
|
||||||
state.rax = len;
|
arch.setSyscallResult(state, len);
|
||||||
} else {
|
} else {
|
||||||
state.rax = @bitCast(@as(i64, -1));
|
arch.setSyscallResult(state, @bitCast(@as(i64, -1)));
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
else => state.rax = @bitCast(@as(i64, -1)),
|
else => arch.setSyscallResult(state, @bitCast(@as(i64, -1))),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -128,7 +128,7 @@ fn syscall(state: *arch.CpuState) void {
|
|||||||
fn resetRecords() void {
|
fn resetRecords() void {
|
||||||
ping_count = 0;
|
ping_count = 0;
|
||||||
write_len = 0;
|
write_len = 0;
|
||||||
write_cs = 0;
|
write_from_user = false;
|
||||||
write_count = 0;
|
write_count = 0;
|
||||||
exit_code = 0;
|
exit_code = 0;
|
||||||
}
|
}
|
||||||
@@ -257,11 +257,11 @@ fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!stru
|
|||||||
return error.BadEntry;
|
return error.BadEntry;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Load one page of a segment into address space `pml4`: a fresh frame, zeroed
|
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
|
||||||
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||||
/// On a later failure the whole address space is torn down, which frees every
|
/// On a later failure the whole address space is torn down, which frees every
|
||||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||||
fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
const dst: [*]u8 = @ptrFromInt(danos.physToVirt(frame));
|
||||||
@memset(dst[0..page_size], 0);
|
@memset(dst[0..page_size], 0);
|
||||||
@@ -270,7 +270,7 @@ fn loadPageInto(pml4: u64, image: []const u8, seg: Segment, page_index: u64) Ini
|
|||||||
const n = @min(page_size, seg.filesz - page_off);
|
const n = @min(page_size, seg.filesz - page_off);
|
||||||
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
|
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
|
||||||
}
|
}
|
||||||
arch.mapUserPageInto(pml4, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
arch.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
/// Load a user ELF image into a fresh address space and spawn it as a scheduled
|
||||||
@@ -286,15 +286,15 @@ pub fn spawnProcess(image: []const u8, priority: u3) InitError!void {
|
|||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
|
||||||
const pml4 = arch.createAddressSpace() orelse return error.OutOfMemory;
|
const aspace = arch.createAddressSpace() orelse return error.OutOfMemory;
|
||||||
errdefer arch.destroyAddressSpace(pml4);
|
errdefer arch.destroyAddressSpace(aspace);
|
||||||
|
|
||||||
for (segs[0..parsed.count]) |seg| {
|
for (segs[0..parsed.count]) |seg| {
|
||||||
for (0..seg.pages()) |i| try loadPageInto(pml4, image, seg, i);
|
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
||||||
}
|
}
|
||||||
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
arch.mapUserPageInto(pml4, stack_virt, stack_frame, true, false); // RW + NX
|
arch.mapUserPageInto(aspace, stack_virt, stack_frame, true, false); // RW + NX
|
||||||
|
|
||||||
if (!sched.spawnUserLocked(pml4, parsed.entry, stack_virt + page_size, priority))
|
if (!sched.spawnUserLocked(aspace, parsed.entry, stack_virt + page_size, priority))
|
||||||
return error.OutOfMemory;
|
return error.OutOfMemory;
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-2
@@ -156,9 +156,9 @@ CASES = [
|
|||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
||||||
# 0x5 (present|user) at the user RIP. ([\s\S] spans lines; `.` doesn't.)
|
# 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.)
|
||||||
{"name": "user-pf",
|
{"name": "user-pf",
|
||||||
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*RIP\s*: 0x00007000000000",
|
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*IP\s*: 0x00007000000000",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# The real user binary: the bootloader ships sbin/init off the ESP, the
|
# The real user binary: the bootloader ships sbin/init off the ESP, the
|
||||||
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
|
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
|
||||||
|
|||||||
Reference in New Issue
Block a user