From 546dd44a2a9b1b2c65949e9334fa396f0feccb62 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 8 Jul 2026 22:15:07 +0100 Subject: [PATCH] isolation M1: ring 3 + a real /sbin/init, end to end MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ring 3 works: user GDT descriptors (sysret-ready layout), TSS.rsp0, U/S-bit user mappings (W^X preserved), an int 0x80 syscall gate with a mutable trap frame, and a setjmp-style enter/exit path. /sbin/init is a real freestanding Zig binary built from sbin/, shipped on the ESP, loaded by the bootloader (BootInfo.init_base/len), validated and mapped by an in-kernel user-ELF loader, and run at CPL 3 — syscalls: exit, ping, write. Tests: user, user-pf (U/S isolation proof, error code 0x5), init. Suite 27/27. Co-Authored-By: Claude Fable 5 --- build.zig | 30 +++ docs/syscall.md | 8 +- docs/vision.md | 11 +- sbin/init.zig | 54 ++++++ sbin/linker.ld | 45 +++++ src/boot/efi.zig | 48 +++++ src/kernel/arch/x86_64/cpu.zig | 47 ++++- src/kernel/arch/x86_64/gdt.zig | 35 +++- src/kernel/arch/x86_64/idt.zig | 25 ++- src/kernel/arch/x86_64/isr.s | 94 ++++++++++ src/kernel/arch/x86_64/paging.zig | 29 +++ src/kernel/arch/x86_64/tss.zig | 24 ++- src/kernel/main.zig | 21 ++- src/kernel/tests.zig | 68 +++++++ src/kernel/usermode.zig | 298 ++++++++++++++++++++++++++++++ src/root.zig | 6 + test/qemu_test.py | 20 ++ 17 files changed, 838 insertions(+), 25 deletions(-) create mode 100644 sbin/init.zig create mode 100644 sbin/linker.ld create mode 100644 src/kernel/usermode.zig diff --git a/build.zig b/build.zig index f177510..5915b56 100644 --- a/build.zig +++ b/build.zig @@ -145,6 +145,31 @@ pub fn build(b: *std.Build) void { b.installArtifact(exe); + // --- /sbin/init: the first user-space program --- + // Its own tiny freestanding binary, linked at a fixed address inside the + // kernel's user region (usermode.zig) and started in ring 3 by the kernel's + // user-ELF loader. `.large` because the image base is above 4 GiB — small/ + // medium code models emit 32-bit absolute relocations that can't reach. + // Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its + // size has no reason to track the kernel's optimize mode. + const init_exe = b.addExecutable(.{ + .name = "init", + .root_module = b.createModule(.{ + .root_source_file = b.path("sbin/init.zig"), + .target = kernel_target, + .optimize = .ReleaseSmall, + .code_model = .large, + .single_threaded = true, + .sanitize_c = .off, + .stack_check = false, + .stack_protector = false, + }), + }); + init_exe.setLinkerScript(b.path("sbin/linker.ld")); + init_exe.entry = .{ .symbol_name = "_start" }; + init_exe.image_base = 0x7000_0000_0000; + b.installArtifact(init_exe); + // Boot methods live in src/boot/, one per way of getting the kernel running. // Each is its own binary/entry (a loader is built for its own target); today // that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis. @@ -203,6 +228,10 @@ pub fn build(b: *std.Build) void { const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "esp" } }, }); + // The bootloader loads init from sbin/init on the same volume. + const init_install = b.addInstallArtifact(init_exe, .{ + .dest_dir = .{ .override = .{ .custom = "esp/sbin" } }, + }); // The firmware needs to write NVRAM, so give it a writable copy of the vars. const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars }); @@ -239,6 +268,7 @@ pub fn build(b: *std.Build) void { run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) }); run_efi.step.dependOn(&efi_install.step); run_efi.step.dependOn(&kernel_install.step); + run_efi.step.dependOn(&init_install.step); const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-.log"); run_efi_step.dependOn(&run_efi.step); diff --git a/docs/syscall.md b/docs/syscall.md index 3c6c57b..eabbf2b 100644 --- a/docs/syscall.md +++ b/docs/syscall.md @@ -1,5 +1,11 @@ # System Calls -System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel). +System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel). + +> **Status:** danos currently has a placeholder M1 surface behind an `int 0x80` +> gate — `0 = exit(code)`, `1 = ping(value)`, `2 = write(ptr, len)` (see +> `src/kernel/usermode.zig`, used by `sbin/init.zig`). It exists to prove the +> ring transition; the microkernel set below replaces it (via `syscall`/`sysret`) +> when processes land (M3). ## The Mechanism of a Syscall diff --git a/docs/vision.md b/docs/vision.md index cc184b4..c08102c 100644 --- a/docs/vision.md +++ b/docs/vision.md @@ -81,11 +81,16 @@ prerequisites. (with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a [heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking, -and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md). +in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), and +**ring 3**: user GDT/TSS plumbing, U/S-bit mappings, an `int 0x80` syscall gate, and +`/sbin/init` — a real user ELF built from `sbin/`, shipped on the boot volume, loaded +by the kernel, run at CPL 3 — plus a [test harness](testing.md). - **Isolation track** — **user mode + address-space isolation** (higher-half kernel, - ring 3, per-process page tables). The substrate everything else needs. *Next, and a - prerequisite for the resilience and driver tracks.* + ring 3, per-process page tables). The substrate everything else needs. *In + progress: ring 3 + a loaded `/sbin/init` work (M1); next the higher-half move (M2), + then per-process address spaces + `syscall`/`sysret` + init as a real schedulable + process (M3), then ELF/initrd generalisation (M4).* - **Resilience track** — fault → kill → notify, a supervisor/reincarnation server, resource cleanup on death, then a restartable driver as proof. Needs isolation. See [resilience.md](resilience.md). diff --git a/sbin/init.zig b/sbin/init.zig new file mode 100644 index 0000000..b5e0bd1 --- /dev/null +++ b/sbin/init.zig @@ -0,0 +1,54 @@ +//! /sbin/init — the first user-space program. Built as its own freestanding +//! binary (see build.zig), shipped on the boot volume at sbin/init, loaded by +//! the bootloader, and started in ring 3 by the kernel's user-ELF loader +//! (src/kernel/usermode.zig). It talks to the kernel only through the +//! `int $0x80` syscall gate. +//! +//! Today it just proves the path — say hello, exit — and grows into the real +//! init (service supervision) once processes are schedulable (M3). + +const std = @import("std"); + +// The M1 syscall numbers (usermode.zig): 0 = exit(code), 2 = write(ptr, len). +const sys_exit = 0; +const sys_write = 2; + +fn syscall2(n: u64, a: u64, b: u64) u64 { + return asm volatile ("int $0x80" + : [ret] "={rax}" (-> u64), + : [n] "{rax}" (n), + [a] "{rdi}" (a), + [b] "{rsi}" (b), + : .{ .memory = true }); +} + +fn write(msg: []const u8) void { + _ = syscall2(sys_write, @intFromPtr(msg.ptr), msg.len); +} + +fn exit(code: u64) noreturn { + _ = syscall2(sys_exit, code, 0); + unreachable; // the kernel never returns from exit +} + +/// Entry. Naked: the kernel enters with rsp 16-aligned, but a SysV function +/// expects rsp ≡ 8 (mod 16) on entry (as if reached by `call`) — so re-enter +/// the ABI with an actual call. The trap after is unreachable. +pub export fn _start() callconv(.naked) noreturn { + asm volatile ( + \\call init_main + \\ud2 + ); +} + +export fn init_main() callconv(.c) noreturn { + write("init: hello from user space\n"); + exit(0); +} + +/// No runtime to unwind into — report the panic as a nonzero exit code. +pub const panic = std.debug.FullPanic(struct { + fn panic(_: []const u8, _: ?usize) noreturn { + exit(127); + } +}.panic); diff --git a/sbin/linker.ld b/sbin/linker.ld new file mode 100644 index 0000000..2b889e2 --- /dev/null +++ b/sbin/linker.ld @@ -0,0 +1,45 @@ +/* /sbin/init link layout. + * + * Linked at a fixed user-space virtual base (set by `image_base` in build.zig, + * inside the kernel's user region). Same discipline as the kernel's script: + * one PT_LOAD per permission set, every section page-aligned, so the kernel's + * user-ELF loader can map each segment with exact W^X permissions. Note the + * linker also emits a read-only PT_LOAD covering the ELF headers at the image + * base, so the entry point comes from e_entry, not the base address. + */ + +ENTRY(_start) + +/* FLAGS bits: 1=X, 2=W, 4=R. */ +PHDRS { + text PT_LOAD FLAGS(5); /* R + X */ + rodata PT_LOAD FLAGS(4); /* R */ + data PT_LOAD FLAGS(6); /* R + W */ +} + +SECTIONS { + .text ALIGN(4K) : { + *(.text .text.*) + } :text + + .rodata ALIGN(4K) : { + *(.rodata .rodata.*) + } :rodata + + .data ALIGN(4K) : { + *(.data .data.*) + } :data + + /* .bss occupies memory but not file space; the loader zeroes the + * filesz..memsz gap. */ + .bss ALIGN(4K) : { + *(.bss .bss.*) + *(COMMON) + } :data + + /DISCARD/ : { + *(.comment) + *(.note .note.*) + *(.eh_frame .eh_frame_hdr) + } +} diff --git a/src/boot/efi.zig b/src/boot/efi.zig index 8f30d1c..490ea6f 100644 --- a/src/boot/efi.zig +++ b/src/boot/efi.zig @@ -11,6 +11,10 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice; /// build.zig). UEFI wants a UTF-16, null-terminated path. const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel"); +/// Path of the init program on the boot volume (UEFI paths use backslashes; +/// the FAT driver walks the components itself, so no directory dance needed). +const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("sbin\\init"); + /// Physical page size, and the sentinel UEFI uses to seek to end-of-file. const page_size = 4096; const seek_end = 0xffff_ffff_ffff_ffff; @@ -54,6 +58,13 @@ fn boot() !noreturn { const entry = try loadKernel(bs, &boot_info); + // Best effort: a volume without sbin/init still boots (kernel-only). + loadInit(bs, &boot_info) catch |err| { + log("danos: no sbin/init ("); + logBytes(@errorName(err)); + log(") - booting without user space\r\n"); + }; + log("danos: kernel loaded, exiting boot services\r\n"); boot_info.memory_map = try exitBootServices(bs); @@ -198,6 +209,43 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize { return loadElf(bs, image, boot_info); } +/// Read the init program (sbin/init) into memory that outlives the loader and +/// record it in the handoff. The pool buffer is deliberately NOT freed: it's +/// LoaderData, which the memory-map conversion classifies as reserved, so the +/// kernel identity-maps it and reads the ELF from there. The kernel does the +/// loading itself (into ring-3 mappings) — the loader just ferries the bytes. +fn loadInit(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void { + const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse + return error.NoLoadedImage; + const device = loaded.device_handle orelse return error.NoBootDevice; + const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse + return error.NoFileSystem; + + const root = try fs.openVolume(); + defer _ = root.close() catch {}; + + const file = try root.open(init_file_name, .read, .{}); + defer _ = file.close() catch {}; + + try file.setPosition(seek_end); + const size: usize = @intCast(try file.getPosition()); + try file.setPosition(0); + if (size == 0) return error.EmptyFile; + + const image = try bs.allocatePool(.loader_data, size); // survives the handoff + + var read_total: usize = 0; + while (read_total < size) { + const n = try file.read(image[read_total..]); + if (n == 0) return error.UnexpectedEof; + read_total += n; + } + + boot_info.init_base = @intFromPtr(image.ptr); + boot_info.init_len = size; + log("danos: sbin/init loaded\r\n"); +} + /// Validate the ELF, copy every PT_LOAD segment to its physical address, and /// record each segment's layout so the kernel can re-map itself with the right /// permissions. diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index 5677c86..40bb985 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -76,6 +76,45 @@ pub fn unmapPage(virt: u64) void { paging.unmap(virt); } +/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps +/// W^X: code read-only + executable, data writable + no-execute. +pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void { + paging.mapUser(virt, phys, writable, executable); +} + +// --- ring 3 entry/exit ----------------------------------------------------- + +/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context, +/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0, +/// so ring-3 interrupts land on a good stack), builds an iretq frame with the +/// user selectors, and iretq's. "Returns" only when the user program triggers +/// the exit path (user_exit_to_kernel). +extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void; + +/// Abandon the in-flight ring-3 trap context and resume the kernel as if +/// `enter_user` had returned (defined in isr.s). Called by the exit syscall. +extern fn user_exit_to_kernel() callconv(.c) noreturn; + +/// Run user code at `rip` with stack `rsp` on this core (`cpu` = the caller's CPU +/// index; the arch layer can't ask the scheduler). Returns after the user program +/// exits via syscall. Interrupts are disabled on return (the exit arrives through +/// an interrupt gate) — the caller re-enables. +pub fn enterUser(cpu: usize, rip: u64, rsp: u64) void { + enter_user(rip, rsp, tss.rsp0Ptr(cpu)); +} + +/// Never returns to the user program: unwind to the kernel context that called +/// `enterUser`. For the exit syscall's handler. +pub fn userExit() noreturn { + user_exit_to_kernel(); +} + +/// Register the handler for the ring-3 syscall gate (int 0x80, vector 128). The +/// handler may write the trap frame (e.g. rax as the return value). +pub fn setSyscallHandler(handler: *const fn (*idt.CpuState) void) void { + idt.setSyscallHandler(handler); +} + /// CR3 holds the physical address of the active top-level page table. pub fn readCr3() u64 { return asm volatile ("mov %%cr3, %[out]" @@ -84,9 +123,11 @@ pub fn readCr3() u64 { } /// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU -/// data pointer (there's no user mode yet, so no `swapgs` dance — GS base is always -/// the running core's per-CPU block). Set once per core during bring-up, after the -/// GDT is loaded (loading a GS *selector* would otherwise clobber this base). +/// data pointer. Because it's always read back via `rdmsr` (never gs-relative +/// addressing), no `swapgs` dance is needed even with user mode: the MSR is +/// privileged, ring 3 can't touch it, and its value is unaffected by ring +/// transitions. Set once per core during bring-up, after the GDT is loaded +/// (loading a GS *selector* would otherwise clobber this base). const ia32_gs_base = 0xC000_0101; /// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each diff --git a/src/kernel/arch/x86_64/gdt.zig b/src/kernel/arch/x86_64/gdt.zig index 7b24f77..d759370 100644 --- a/src/kernel/arch/x86_64/gdt.zig +++ b/src/kernel/arch/x86_64/gdt.zig @@ -1,8 +1,8 @@ //! Global Descriptor Table. In long mode segmentation is mostly vestigial, but //! the CPU still needs valid code/data segment descriptors, and the IDT's gates //! reference a code selector — so we install our own flat GDT with known -//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever -//! the firmware left in place. +//! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3) +//! rather than trusting whatever the firmware left in place. //! //! The code/data descriptors are identical on every core, but the **TSS descriptor //! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see @@ -14,20 +14,35 @@ const config = @import("config"); /// Selectors into the table (index * 8). Same on every core's GDT. pub const kernel_code = 0x08; pub const kernel_data = 0x10; -pub const tss_selector = 0x18; +pub const user_data = 0x18; +pub const user_code = 0x20; +pub const tss_selector = 0x28; + +/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data* +/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS +/// raises #GP(0). +pub const user_code_rpl3 = user_code | 3; +pub const user_data_rpl3 = user_data | 3; const max_cpus = config.max_cpus; -const entries = 5; // null, code, data, TSS-low, TSS-high +const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high -/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor, +/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor, /// filled in per core by `setTssFor`. -/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF -/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF +/// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF +/// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF +/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF +/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF +/// User data sits below user code so a future SYSRET works unchanged: it loads +/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10 +/// those land on 0x20 (user code) and 0x18 (user data). const template = [entries]u64{ 0, // null descriptor (required) 0x00AF9A000000FFFF, // kernel code (0x08) 0x00CF92000000FFFF, // kernel data (0x10) - 0, // TSS descriptor low (0x18) + 0x00CFF2000000FFFF, // user data (0x18) + 0x00AFFA000000FFFF, // user code (0x20) + 0, // TSS descriptor low (0x28) 0, // TSS descriptor high }; @@ -38,13 +53,13 @@ var gdts = [_][entries]u64{template} ** max_cpus; /// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit /// TSS. Write it into that core's GDT before it loads the TSS selector. pub fn setTssFor(cpu: usize, base: u64, limit: u64) void { - gdts[cpu][3] = (limit & 0xFFFF) | + gdts[cpu][5] = (limit & 0xFFFF) | ((base & 0xFFFF) << 16) | (((base >> 16) & 0xFF) << 32) | (@as(u64, 0x89) << 40) | (((limit >> 16) & 0xF) << 48) | (((base >> 24) & 0xFF) << 56); - gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF; + gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF; } /// The operand `lgdt` wants: table byte-length minus one, then its address. diff --git a/src/kernel/arch/x86_64/idt.zig b/src/kernel/arch/x86_64/idt.zig index da5b02a..f9fff47 100644 --- a/src/kernel/arch/x86_64/idt.zig +++ b/src/kernel/arch/x86_64/idt.zig @@ -26,6 +26,19 @@ pub fn setHandler(vector: usize, handler: Handler) void { handlers[vector] = handler; } +/// The ring-3 syscall gate's vector (`int $0x80`, the classic choice — well away +/// from the device range) and its handler. Unlike device handlers, a syscall +/// handler gets the (mutable) trap frame: it reads its arguments from the saved +/// user registers and writes rax as the return value, which isr_common then +/// restores into the user context. +pub const syscall_vector = 128; + +var syscall_handler: ?*const fn (*CpuState) void = null; + +pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void { + syscall_handler = handler; +} + /// The register + trap frame the ISR stubs build on the stack, laid out so the /// lowest address (where RSP points when we call the handler) is the first field. /// See the push order in `isrCommon` below. @@ -128,6 +141,13 @@ pub fn init() void { // Run the double-fault handler (vector 8) on IST1: a #DF usually means the // current stack is unusable, so it needs a guaranteed-good one. See tss.zig. idt[8].ist = tss.double_fault_ist; + // The syscall gate. Installed outside the 0..gate_count loop (stubs 48-127 + // don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a + // #GP. An interrupt gate (not trap): IF is cleared for the handler, which + // the ring-3 exit path relies on. + const syscall_stub = @extern(*const anyopaque, .{ .name = "isr128" }); + setGate(syscall_vector, @intFromPtr(syscall_stub)); + idt[syscall_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate loadOnThisCpu(); } @@ -145,9 +165,12 @@ pub fn loadOnThisCpu() void { /// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the /// assembly stubs can `call` it by name. Exceptions are terminal; device /// interrupts run their handler, get acknowledged, and return. -export fn interruptDispatch(state: *const CpuState) callconv(.c) void { +export fn interruptDispatch(state: *CpuState) callconv(.c) void { if (state.vector < 32) { on_fault(state); // CPU exception — never returns + } else if (state.vector == syscall_vector) { + // Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI. + if (syscall_handler) |handler| handler(state); } else if (handlers[state.vector]) |handler| { // Acknowledge before running the handler: a handler that switches tasks // (the scheduler) may not return promptly, and the LAPIC mustn't wait on diff --git a/src/kernel/arch/x86_64/isr.s b/src/kernel/arch/x86_64/isr.s index 0b43191..3d8c050 100644 --- a/src/kernel/arch/x86_64/isr.s +++ b/src/kernel/arch/x86_64/isr.s @@ -76,6 +76,96 @@ task_trampoline: 1: hlt # if the entry returns, idle (still preemptible) jmp 1b +# --- ring 3 entry/exit ------------------------------------------------------ + +# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0) +# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer +# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as +# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames +# grow safely into it — then builds the 5-word iretq frame with the user +# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the +# timer keeps running in user mode. +.global enter_user +enter_user: + push %rbx + push %rbp + push %r12 + push %r13 + push %r14 + push %r15 + mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to + mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here + push $0x1B # user SS (0x18 | RPL 3) + push %rsi # user RSP + push $0x202 # RFLAGS: IF | reserved-1 + push $0x23 # user CS (0x20 | RPL 3) + push %rdi # user RIP + iretq + +# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the +# dead zone below user_saved_rsp) and return as if enter_user's call completed. +# Reached from the exit syscall's handler; interrupts are off (interrupt gate) +# and stay off — the Zig caller re-enables. +.global user_exit_to_kernel +user_exit_to_kernel: + mov user_saved_rsp(%rip), %rsp + pop %r15 + pop %r14 + pop %r13 + pop %r12 + pop %rbp + pop %rbx + ret + +.section .bss +.balign 8 +user_saved_rsp: + .skip 8 +.text + +# --- user-mode test programs ------------------------------------------------- +# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and +# entered via enter_user. Position-independent (immediates and short jumps only). +# In .rodata: these bytes are data to the kernel — they only execute at CPL 3 +# from the user mapping. +.section .rodata + +# The hello program: ping syscall (rax=1) with a value, a delay loop long enough +# for several timer ticks to land while at CPL 3 (proving interrupt-from-user + +# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety +# net in case exit ever returns. +.global user_prog_start +.global user_prog_end +user_prog_start: + mov $1, %rax + mov $0xC0DE, %rdi + int $0x80 + mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG +1: dec %rcx + jnz 1b + mov $1, %rax + mov $0xBEEF, %rdi + int $0x80 + mov $0, %rax + int $0x80 +2: jmp 2b +user_prog_end: + +# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC +# page is unconditionally mapped supervisor (paging.zig), so this must take a +# #PF with error code 0x5 (present | user) before any access happens. The +# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax` +# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page. +.global user_pf_start +.global user_pf_end +user_pf_start: + mov $0xFEE00000, %ecx + mov (%rcx), %rax +1: jmp 1b +user_pf_end: + +.text + # Stub for a vector the CPU does NOT push an error code for: push a dummy 0. .macro STUB_NOERR vec .global isr\vec @@ -145,6 +235,10 @@ STUB_NOERR 45 STUB_NOERR 46 STUB_NOERR 47 +# The syscall gate (int $0x80 from ring 3). Same frame shape as every other +# vector; dispatched specially in interruptDispatch. +STUB_NOERR 128 + .extern interruptDispatch # Shared tail. Register push order here defines the CpuState field order. diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index 50ef987..c8d5bdf 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -18,6 +18,7 @@ const page_size = danos.page_size; // Page-table entry bits. const present: u64 = 1 << 0; const writable: u64 = 1 << 1; +const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level) const no_execute: u64 = 1 << 63; const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; @@ -128,6 +129,34 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void { invalidate(virt); } +/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or +/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves +/// existing entries untouched. Only used under user-exclusive virtual ranges, so +/// no kernel mapping's protection is widened (the leaf still governs). +fn descendUser(entry: *u64) u64 { + const table = descend(entry); + entry.* |= user; + return table; +} + +/// Map one 4 KiB page `virt` -> `phys` accessible from ring 3. W^X is the +/// caller's contract: code pages are read-only + executable, data pages are +/// writable + no-execute. `virt` must lie in a user-exclusive region (see +/// `descendUser`). +pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void { + var flags: u64 = present | user; + if (writable_page) flags |= writable; + if (!executable) flags |= no_execute; + const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF]; + const pdpt = descendUser(pml4e); + const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF]; + const pd = descendUser(pdpte); + const pde = &tableAt(pd)[(virt >> 21) & 0x1FF]; + const pt = descendUser(pde); + tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags; + invalidate(virt); +} + /// Whether `virt` is currently mapped **executable** — present with the NX bit /// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page /// case). Returns false if unmapped. Used for W^X checks in tests. diff --git a/src/kernel/arch/x86_64/tss.zig b/src/kernel/arch/x86_64/tss.zig index 076db3a..47b8e32 100644 --- a/src/kernel/arch/x86_64/tss.zig +++ b/src/kernel/arch/x86_64/tss.zig @@ -1,9 +1,11 @@ -//! Task State Segment and its interrupt stack. In long mode the TSS's main job -//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU -//! switches to that stack when the exception fires — no matter how broken the -//! interrupted stack was. We use IST1 for the double-fault handler, so a fault -//! that happens *because* the current stack is unusable still lands on solid -//! ground instead of triple-faulting. +//! Task State Segment and its interrupt stacks. In long mode the TSS has two +//! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry, +//! and the CPU switches to that stack when the exception fires — no matter how +//! broken the interrupted stack was. We use IST1 for the double-fault handler, +//! so a fault that happens *because* the current stack is unusable still lands +//! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack +//! the CPU switches to when an interrupt arrives from ring 3 (published by the +//! user-mode entry path via `rsp0Ptr`). //! //! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at //! once can't share one fault stack. So the TSS and its IST stack are per-core, @@ -50,6 +52,16 @@ var ap_ist_top = [_]usize{0} ** max_cpus; // per-AP IST stack top (0 = BSP / not /// Loads the task register with the TSS selector. Defined in isr.s. extern fn load_tr(selector: u16) callconv(.c) void; +/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a +/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset, +/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a +/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path +/// (enter_user in isr.s) writes the current kernel stack pointer through this +/// before dropping to user mode. +pub fn rsp0Ptr(cpu: usize) *align(4) u64 { + return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4); +} + /// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the /// BSP before waking that core; read by the core's own `setupThisCpu`. pub fn setApIstStack(cpu: usize, top: usize) void { diff --git a/src/kernel/main.zig b/src/kernel/main.zig index ea284d3..b4a16e7 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -7,6 +7,7 @@ const log = @import("log.zig"); const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const scheduler = @import("scheduler.zig"); +const usermode = @import("usermode.zig"); const platform = @import("platform"); const tests = @import("tests.zig"); const build_options = @import("build_options"); @@ -245,7 +246,25 @@ fn kmain(boot_info: *const BootInfo) noreturn { log.checkpoint(cp_running); status("kernel initialised.\n"); - // TODO: init process + // Hand over to user space: run /sbin/init (read off the boot volume by the + // loader) in ring 3. Preemption is off for the run — the M1 user-mode path + // publishes *this* core's TSS.rsp0 and must not migrate (the flag is + // global, so the system goes cooperative meanwhile; the other cores are + // idle). init becomes a real schedulable process in M3. + if (boot_info.init_len != 0) { + status("starting /sbin/init...\n"); + const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len]; + scheduler.setPreemption(false); + const code = usermode.runInitElf(image); + scheduler.setPreemption(true); + if (code) |c| { + statusPrint("/sbin/init exited with code {d}.\n", .{c}); + } else |err| { + statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)}); + } + } else { + status("no /sbin/init on the boot volume.\n"); + } status("\nnothing left to do; halting CPU.\n"); diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index ba225e4..c31aabd 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -17,6 +17,7 @@ const pmm = @import("pmm.zig"); const heap = @import("heap.zig"); const sched = @import("scheduler.zig"); const ipc = @import("ipc.zig"); +const usermode = @import("usermode.zig"); /// Formatted write straight to serial, independent of the framebuffer console. fn log(comptime fmt: []const u8, args: anytype) void { @@ -92,6 +93,12 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { faultNoExecute(); } else if (eql(case, "fault-null")) { faultNull(); + } else if (eql(case, "user")) { + userTest(); + } else if (eql(case, "user-pf")) { + userPfTest(); + } else if (eql(case, "init")) { + initTest(boot_info); } else if (eql(case, "poweroff")) { powerTest(.off); } else if (eql(case, "reboot")) { @@ -702,6 +709,67 @@ fn smpRetryTest() void { result(); } +// --- ring 3 (user mode) ----------------------------------------------------- + +/// The full ring-3 round trip: enter user mode, take syscalls and timer +/// interrupts from CPL 3, and come back. Preemption is disabled for the run — +/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate +/// (interrupts still fire and iretq back into ring 3, which is the point). +fn userTest() void { + log("DANOS-TEST-BEGIN: user\n", .{}); + sched.setPreemption(false); + const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: { + log("DANOS-USER: run failed: {s}\n", .{@errorName(err)}); + break :blk false; + }; + sched.setPreemption(true); + + check("user program ran and exited (ring-3 round trip)", ran); + check("two ping syscalls received", usermode.ping_count == 2); + check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF); + check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23); + check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks); + result(); +} + +/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present, +/// supervisor) must page-fault with error code 0x5 (present | user) at the user +/// RIP. The fault report is the pass signal (matched by the harness); if the +/// read is somehow allowed the blob spins and the harness times out. +fn userPfTest() void { + log("DANOS-TEST-BEGIN: user-pf\n", .{}); + sched.setPreemption(false); + _ = usermode.run(usermode.pfBlob()) catch {}; + log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{}); +} + +/// The full user-binary path: the bootloader read sbin/init off the boot +/// volume and handed it over; load it as a user ELF and run it in ring 3. The +/// same call the normal boot path makes — here with teeth. +fn initTest(boot_info: *const BootInfo) void { + log("DANOS-TEST-BEGIN: init\n", .{}); + check("bootloader handed over sbin/init", boot_info.init_len != 0); + if (boot_info.init_len == 0) { + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len]; + sched.setPreemption(false); // see userTest: pins the run to this core's rsp0 + const code = usermode.runInitElf(image); + sched.setPreemption(true); + + if (code) |c| { + check("init loaded, ran, and exited (user ELF path)", true); + check("init exited cleanly (code 0)", c == 0); + } else |err| { + log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)}); + check("init loaded, ran, and exited (user ELF path)", false); + } + check("init's write arrived intact", eql(usermode.write_buf[0..usermode.write_len], "init: hello from user space\n")); + check("write came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23); + result(); +} + fn faultInvalidOpcode() void { log("DANOS-TEST-BEGIN: fault-ud\n", .{}); asm volatile ("ud2"); diff --git a/src/kernel/usermode.zig b/src/kernel/usermode.zig new file mode 100644 index 0000000..0294436 --- /dev/null +++ b/src/kernel/usermode.zig @@ -0,0 +1,298 @@ +//! Ring-3 execution — the isolation track's first rung (M1). Two entry points: +//! `run` executes a raw code blob (the user/user-pf test programs), and +//! `runInitElf` loads and runs a real user ELF (`/sbin/init`, handed over by +//! the bootloader). Both run at CPL 3 in the current (global) page tables: +//! frames are mapped user-accessible with W^X (code RO+X, data RW+NX), the CPU +//! drops privilege via iretq, and the program talks to the kernel only through +//! the `int $0x80` syscall gate. No processes or per-task address spaces yet — +//! this proves the privilege mechanisms (user descriptors, TSS.rsp0, the U/S +//! page bit, ring transitions) that M3 builds on. +//! +//! Caller contract (M1 limitations): +//! - enter/exit publishes TSS.rsp0 on the *current* core only, so the caller +//! must prevent migration for the duration — disable preemption around the +//! call (rsp0-per-context-switch arrives with real user tasks in M3). +//! - The kernel-side saved context (`user_saved_rsp` in isr.s) is a single +//! global: at most one core may be inside user mode at a time. + +const std = @import("std"); +const elf = std.elf; +const danos = @import("danos"); +const arch = @import("arch"); +const pmm = @import("pmm.zig"); +const sched = @import("scheduler.zig"); +const log = @import("log.zig"); + +const page_size = danos.page_size; + +/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from +/// the identity map (low indices) and the vmm test address (index 128), so +/// setting the U/S bit on its intermediate tables widens no kernel mapping. +/// An ELF image may occupy [code_virt, stack_virt); the stack page sits above. +pub const code_virt: u64 = 0x0000_7000_0000_0000; +pub const stack_virt: u64 = 0x0000_7000_0020_0000; + +// The hand-assembled user program blobs (isr.s, .rodata). +const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" }); +const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" }); +const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" }); +const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" }); + +/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit. +pub fn helloBlob() []const u8 { + return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)]; +} + +/// The isolation-proof program: reads a kernel-only page, must #PF. +pub fn pfBlob() []const u8 { + return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)]; +} + +/// What a ping syscall recorded — evidence for the test to assert on. +pub const Ping = struct { value: u64 = 0, cs: u64 = 0, ticks: u64 = 0 }; +pub var pings = [_]Ping{.{}} ** 2; +pub var ping_count: usize = 0; + +/// What write syscalls produced (accumulated), and the exit syscall's code. +pub var write_buf: [256]u8 = undefined; +pub var write_len: usize = 0; +pub var write_cs: u64 = 0; +pub var exit_code: u64 = 0; + +/// The M1 syscall surface, dispatched on the saved user rax: +/// 0 = exit(code) — unwind back into the kernel context that entered +/// 1 = ping(value) — record rdi + the caller's CS + the current tick count +/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT: +/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md) +/// replaces this in M3; rax is written back as the return value already, since +/// isr_common restores user registers from the trap frame. +fn syscall(state: *arch.CpuState) void { + switch (state.rax) { + 0 => { + exit_code = state.rdi; + arch.userExit(); + }, + 1 => { + if (ping_count < pings.len) { + pings[ping_count] = .{ .value = state.rdi, .cs = state.cs, .ticks = arch.ticks() }; + ping_count += 1; + } + state.rax = 0; + }, + 2 => { + // The pointer must lie inside the user region (image + stack) — + // kernel addresses and non-canonical values fall outside it, so the + // kernel-side read below can't be steered at kernel data. Length is + // checked first so the upper-bound subtraction can't underflow. + // Known gap (fine for a trusted init): a pointer into an *unmapped* + // hole in the region passes the check and the read #PFs -> on_fault + // halts — a self-DoS, not an isolation break. Copy-in with fault + // recovery is M3+ (with SMAP, once there's a reason to enable it). + const ptr = state.rdi; + const len = state.rsi; + if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) { + const src: [*]const u8 = @ptrFromInt(ptr); + const n = @min(len, write_buf.len - write_len); + @memcpy(write_buf[write_len..][0..n], src[0..n]); + write_len += n; + write_cs = state.cs; + log.write("DANOS-INIT: "); + log.write(src[0..len]); + state.rax = len; + } else { + state.rax = @bitCast(@as(i64, -1)); + } + }, + else => state.rax = @bitCast(@as(i64, -1)), + } +} + +/// Reset the recorded syscall evidence before a user-mode run. +fn resetRecords() void { + ping_count = 0; + write_len = 0; + write_cs = 0; + exit_code = 0; +} + +pub const RunError = error{ ProgramTooBig, OutOfMemory }; + +/// Map `blob` at code_virt with a fresh user stack, drop to ring 3, and return +/// once the program exits via syscall 0. See the migration caveat in the module +/// doc. A program that faults instead never returns (on_fault halts the core). +pub fn run(blob: []const u8) RunError!void { + if (blob.len > page_size) return error.ProgramTooBig; + const code_frame = pmm.alloc() orelse return error.OutOfMemory; + const stack_frame = pmm.alloc() orelse { + pmm.free(code_frame); + return error.OutOfMemory; + }; + + // Fill the code frame through its identity mapping (supervisor RW): the + // user-facing mapping is read-only, and this also sidesteps CR0.WP/SMAP. + // The tail is padded with int3 so a stray jump traps instead of sliding. + const code: [*]u8 = @ptrFromInt(code_frame); + @memcpy(code[0..blob.len], blob); + @memset(code[blob.len..page_size], 0xCC); + + arch.setSyscallHandler(syscall); + arch.mapUserPage(code_virt, code_frame, false, true); // RO + X + arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX + resetRecords(); + + arch.enterUser(sched.currentCpuIndex(), code_virt, stack_virt + page_size); + + // Back via the exit syscall; the interrupt gate left IF clear. + arch.enableInterrupts(); + arch.unmapPage(code_virt); + arch.unmapPage(stack_virt); + pmm.free(code_frame); + pmm.free(stack_frame); +} + +// --- user ELF loading (/sbin/init) ------------------------------------------ + +pub const InitError = error{ + BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds) + BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping + BadEntry, // e_entry not inside an executable segment + ProgramTooBig, // more pages than the loader's budget + OutOfMemory, +}; + +const max_segments = 16; +const max_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway + +const Segment = struct { + vaddr: u64, + memsz: u64, + filesz: u64, + off: u64, + writable: bool, + executable: bool, + + fn pages(self: Segment) u64 { + return (self.memsz + page_size - 1) / page_size; + } +}; + +/// Pages mapped for the running init image, for rollback if loading fails +/// midway. On success the mappings stay for the system's life — there's no +/// address-space teardown until real processes exist (M3). +var loaded = [_]struct { virt: u64, frame: u64 }{.{ .virt = 0, .frame = 0 }} ** (max_pages + 1); +var loaded_count: usize = 0; + +/// Parse and validate every PT_LOAD before touching memory. Bounds are checked +/// against the image and the user region; segments must be page-aligned, +/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only +/// segment, mapped RO+NX). +fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!struct { count: usize, entry: u64 } { + if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf; + const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]); + if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf; + if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf; + if (ehdr.e_machine != .X86_64) return error.BadElf; + if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation + if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf; + if (ehdr.e_phnum > max_segments) return error.BadElf; + const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize; + if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf; + + var count: usize = 0; + var total_pages: u64 = 0; + for (0..ehdr.e_phnum) |i| { + const off = ehdr.e_phoff + i * ehdr.e_phentsize; + const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]); + if (phdr.p_type != elf.PT_LOAD) continue; + if (phdr.p_memsz == 0) continue; + + if (phdr.p_vaddr % page_size != 0) return error.BadSegment; + if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment; + if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment; + // Inside the user image region, strictly below the stack page. + if (phdr.p_vaddr < code_virt) return error.BadSegment; + if (phdr.p_memsz > stack_virt - phdr.p_vaddr) return error.BadSegment; + + const w = phdr.p_flags & elf.PF_W != 0; + const x = phdr.p_flags & elf.PF_X != 0; + if (w and x) return error.BadSegment; // W^X, even for init + + const seg = Segment{ + .vaddr = phdr.p_vaddr, + .memsz = phdr.p_memsz, + .filesz = phdr.p_filesz, + .off = phdr.p_offset, + .writable = w, + .executable = x, + }; + // No overlap with any earlier segment (page-granular, since mapping is). + for (segs[0..count]) |other| { + const a_end = seg.vaddr + seg.pages() * page_size; + const b_end = other.vaddr + other.pages() * page_size; + if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment; + } + total_pages += seg.pages(); + if (total_pages > max_pages) return error.ProgramTooBig; + segs[count] = seg; + count += 1; + } + if (count == 0) return error.BadElf; + + // The entry point must land inside an executable segment. + for (segs[0..count]) |seg| { + if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz) + return .{ .count = count, .entry = ehdr.e_entry }; + } + return error.BadEntry; +} + +/// Map one page of a segment: fresh frame, zero + copy through the identity +/// mapping (the user mapping may be read-only), then the user-visible mapping +/// with the segment's W^X permissions. Records the page for rollback. +fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void { + const frame = pmm.alloc() orelse return error.OutOfMemory; + const dst: [*]u8 = @ptrFromInt(frame); + @memset(dst[0..page_size], 0); + const page_off = page_index * page_size; + if (page_off < seg.filesz) { + const n = @min(page_size, seg.filesz - page_off); + @memcpy(dst[0..n], image[seg.off + page_off ..][0..n]); + } + const virt = seg.vaddr + page_off; + arch.mapUserPage(virt, frame, seg.writable, seg.executable); + loaded[loaded_count] = .{ .virt = virt, .frame = frame }; + loaded_count += 1; +} + +fn unloadAll() void { + for (loaded[0..loaded_count]) |p| { + arch.unmapPage(p.virt); + pmm.free(p.frame); + } + loaded_count = 0; +} + +/// Load a user ELF image, run it in ring 3 from its entry point, and return its +/// exit code. Same caller contract as `run` (preemption off, one core). On +/// success the user mappings are left in place — teardown comes with real +/// processes (M3); on a loading error everything is rolled back. +pub fn runInitElf(image: []const u8) InitError!u64 { + var segs: [max_segments]Segment = undefined; + const parsed = try parseSegments(image, &segs); + + loaded_count = 0; + errdefer unloadAll(); + for (segs[0..parsed.count]) |seg| { + for (0..seg.pages()) |i| try loadPage(image, seg, i); + } + const stack_frame = pmm.alloc() orelse return error.OutOfMemory; + arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX + loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame }; + loaded_count += 1; + + arch.setSyscallHandler(syscall); + resetRecords(); + arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size); + arch.enableInterrupts(); // the exit arrived through an interrupt gate + return exit_code; +} diff --git a/src/root.zig b/src/root.zig index 25924e0..9c52e76 100644 --- a/src/root.zig +++ b/src/root.zig @@ -109,4 +109,10 @@ pub const BootInfo = extern struct { /// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob` /// field instead, so the kernel discovers devices without knowing what booted it. acpi_rsdp: u64 = 0, + /// The raw `/sbin/init` ELF image, read off the boot volume by the loader + /// into memory that survives the handoff (classified reserved, so the kernel + /// identity-maps it and never allocates over it). 0/0 = no init found — the + /// kernel boots without user space. Grows into a full initrd handoff later. + init_base: u64 = 0, + init_len: u64 = 0, }; diff --git a/test/qemu_test.py b/test/qemu_test.py index 10f0db3..dd991d3 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -57,6 +57,8 @@ ARCHES = { ], "efi_app": ("EFI/BOOT/BOOTX64.efi", "BOOTX64.efi"), # (dest in ESP, name in zig-out/bin) "kernel": ("kernel", "kernel"), + # Further files shipped on the ESP: the init user program. + "extra": [("sbin/init", "init")], # Built as a function so we can splice in per-run paths. "qemu_args": lambda a, esp, vars_fd, serial: [ "-machine", "q35", "-m", "128M", @@ -148,6 +150,21 @@ CASES = [ "expect": r"page fault \(vector 14\)", "fail": r"NX not enforced"}, {"name": "fault-null", "expect": r"page fault \(vector 14\)"}, + # Ring 3: a user program runs at CPL 3, makes int 0x80 syscalls, survives + # timer interrupts, and exits back into the kernel. + {"name": "user", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Isolation: a ring-3 read of a kernel-only page must #PF with error code + # 0x5 (present|user) at the user RIP. ([\s\S] spans lines; `.` doesn't.) + {"name": "user-pf", + "expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*RIP\s*: 0x00007000000000", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + # The real user binary: the bootloader ships sbin/init off the ESP, the + # kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly. + {"name": "init", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the # pre-transition marker; the FAIL line only appears if the transition didn't take. {"name": "poweroff", @@ -179,6 +196,9 @@ def make_esp(arch): os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True) shutil.copy(os.path.join(REPO, "zig-out", "bin", efi_name), os.path.join(esp, efi_dest)) shutil.copy(os.path.join(REPO, "zig-out", "bin", kern_name), os.path.join(esp, kern_dest)) + for dest, name in arch.get("extra", []): + os.makedirs(os.path.join(esp, os.path.dirname(dest)), exist_ok=True) + shutil.copy(os.path.join(REPO, "zig-out", "bin", name), os.path.join(esp, dest)) return esp