isolation M1: ring 3 + a real /sbin/init, end to end

Ring 3 works: user GDT descriptors (sysret-ready layout), TSS.rsp0,
U/S-bit user mappings (W^X preserved), an int 0x80 syscall gate with a
mutable trap frame, and a setjmp-style enter/exit path. /sbin/init is a
real freestanding Zig binary built from sbin/, shipped on the ESP,
loaded by the bootloader (BootInfo.init_base/len), validated and mapped
by an in-kernel user-ELF loader, and run at CPL 3 — syscalls: exit,
ping, write. Tests: user, user-pf (U/S isolation proof, error code
0x5), init. Suite 27/27.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-08 22:15:07 +01:00
co-authored by Claude Fable 5
parent 7501bd1703
commit 546dd44a2a
17 changed files with 838 additions and 25 deletions
+48
View File
@@ -11,6 +11,10 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
/// build.zig). UEFI wants a UTF-16, null-terminated path.
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
/// Path of the init program on the boot volume (UEFI paths use backslashes;
/// the FAT driver walks the components itself, so no directory dance needed).
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("sbin\\init");
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
const page_size = 4096;
const seek_end = 0xffff_ffff_ffff_ffff;
@@ -54,6 +58,13 @@ fn boot() !noreturn {
const entry = try loadKernel(bs, &boot_info);
// Best effort: a volume without sbin/init still boots (kernel-only).
loadInit(bs, &boot_info) catch |err| {
log("danos: no sbin/init (");
logBytes(@errorName(err));
log(") - booting without user space\r\n");
};
log("danos: kernel loaded, exiting boot services\r\n");
boot_info.memory_map = try exitBootServices(bs);
@@ -198,6 +209,43 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
return loadElf(bs, image, boot_info);
}
/// Read the init program (sbin/init) into memory that outlives the loader and
/// record it in the handoff. The pool buffer is deliberately NOT freed: it's
/// LoaderData, which the memory-map conversion classifies as reserved, so the
/// kernel identity-maps it and reads the ELF from there. The kernel does the
/// loading itself (into ring-3 mappings) — the loader just ferries the bytes.
fn loadInit(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void {
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
return error.NoLoadedImage;
const device = loaded.device_handle orelse return error.NoBootDevice;
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
return error.NoFileSystem;
const root = try fs.openVolume();
defer _ = root.close() catch {};
const file = try root.open(init_file_name, .read, .{});
defer _ = file.close() catch {};
try file.setPosition(seek_end);
const size: usize = @intCast(try file.getPosition());
try file.setPosition(0);
if (size == 0) return error.EmptyFile;
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
var read_total: usize = 0;
while (read_total < size) {
const n = try file.read(image[read_total..]);
if (n == 0) return error.UnexpectedEof;
read_total += n;
}
boot_info.init_base = @intFromPtr(image.ptr);
boot_info.init_len = size;
log("danos: sbin/init loaded\r\n");
}
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
/// record each segment's layout so the kernel can re-map itself with the right
/// permissions.
+44 -3
View File
@@ -76,6 +76,45 @@ pub fn unmapPage(virt: u64) void {
paging.unmap(virt);
}
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
/// W^X: code read-only + executable, data writable + no-execute.
pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void {
paging.mapUser(virt, phys, writable, executable);
}
// --- ring 3 entry/exit -----------------------------------------------------
/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context,
/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0,
/// so ring-3 interrupts land on a good stack), builds an iretq frame with the
/// user selectors, and iretq's. "Returns" only when the user program triggers
/// the exit path (user_exit_to_kernel).
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
/// `enter_user` had returned (defined in isr.s). Called by the exit syscall.
extern fn user_exit_to_kernel() callconv(.c) noreturn;
/// Run user code at `rip` with stack `rsp` on this core (`cpu` = the caller's CPU
/// index; the arch layer can't ask the scheduler). Returns after the user program
/// exits via syscall. Interrupts are disabled on return (the exit arrives through
/// an interrupt gate) — the caller re-enables.
pub fn enterUser(cpu: usize, rip: u64, rsp: u64) void {
enter_user(rip, rsp, tss.rsp0Ptr(cpu));
}
/// Never returns to the user program: unwind to the kernel context that called
/// `enterUser`. For the exit syscall's handler.
pub fn userExit() noreturn {
user_exit_to_kernel();
}
/// Register the handler for the ring-3 syscall gate (int 0x80, vector 128). The
/// handler may write the trap frame (e.g. rax as the return value).
pub fn setSyscallHandler(handler: *const fn (*idt.CpuState) void) void {
idt.setSyscallHandler(handler);
}
/// CR3 holds the physical address of the active top-level page table.
pub fn readCr3() u64 {
return asm volatile ("mov %%cr3, %[out]"
@@ -84,9 +123,11 @@ pub fn readCr3() u64 {
}
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
/// data pointer (there's no user mode yet, so no `swapgs` dance — GS base is always
/// the running core's per-CPU block). Set once per core during bring-up, after the
/// GDT is loaded (loading a GS *selector* would otherwise clobber this base).
/// data pointer. Because it's always read back via `rdmsr` (never gs-relative
/// addressing), no `swapgs` dance is needed even with user mode: the MSR is
/// privileged, ring 3 can't touch it, and its value is unaffected by ring
/// transitions. Set once per core during bring-up, after the GDT is loaded
/// (loading a GS *selector* would otherwise clobber this base).
const ia32_gs_base = 0xC000_0101;
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
+25 -10
View File
@@ -1,8 +1,8 @@
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
//! reference a code selector — so we install our own flat GDT with known
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
//! the firmware left in place.
//! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3)
//! rather than trusting whatever the firmware left in place.
//!
//! The code/data descriptors are identical on every core, but the **TSS descriptor
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
@@ -14,20 +14,35 @@ const config = @import("config");
/// Selectors into the table (index * 8). Same on every core's GDT.
pub const kernel_code = 0x08;
pub const kernel_data = 0x10;
pub const tss_selector = 0x18;
pub const user_data = 0x18;
pub const user_code = 0x20;
pub const tss_selector = 0x28;
/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data*
/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS
/// raises #GP(0).
pub const user_code_rpl3 = user_code | 3;
pub const user_data_rpl3 = user_data | 3;
const max_cpus = config.max_cpus;
const entries = 5; // null, code, data, TSS-low, TSS-high
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor,
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
/// filled in per core by `setTssFor`.
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF
/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF
/// User data sits below user code so a future SYSRET works unchanged: it loads
/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10
/// those land on 0x20 (user code) and 0x18 (user data).
const template = [entries]u64{
0, // null descriptor (required)
0x00AF9A000000FFFF, // kernel code (0x08)
0x00CF92000000FFFF, // kernel data (0x10)
0, // TSS descriptor low (0x18)
0x00CFF2000000FFFF, // user data (0x18)
0x00AFFA000000FFFF, // user code (0x20)
0, // TSS descriptor low (0x28)
0, // TSS descriptor high
};
@@ -38,13 +53,13 @@ var gdts = [_][entries]u64{template} ** max_cpus;
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
/// TSS. Write it into that core's GDT before it loads the TSS selector.
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
gdts[cpu][3] = (limit & 0xFFFF) |
gdts[cpu][5] = (limit & 0xFFFF) |
((base & 0xFFFF) << 16) |
(((base >> 16) & 0xFF) << 32) |
(@as(u64, 0x89) << 40) |
(((limit >> 16) & 0xF) << 48) |
(((base >> 24) & 0xFF) << 56);
gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF;
gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF;
}
/// The operand `lgdt` wants: table byte-length minus one, then its address.
+24 -1
View File
@@ -26,6 +26,19 @@ pub fn setHandler(vector: usize, handler: Handler) void {
handlers[vector] = handler;
}
/// The ring-3 syscall gate's vector (`int $0x80`, the classic choice — well away
/// from the device range) and its handler. Unlike device handlers, a syscall
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
/// user registers and writes rax as the return value, which isr_common then
/// restores into the user context.
pub const syscall_vector = 128;
var syscall_handler: ?*const fn (*CpuState) void = null;
pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void {
syscall_handler = handler;
}
/// The register + trap frame the ISR stubs build on the stack, laid out so the
/// lowest address (where RSP points when we call the handler) is the first field.
/// See the push order in `isrCommon` below.
@@ -128,6 +141,13 @@ pub fn init() void {
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
idt[8].ist = tss.double_fault_ist;
// The syscall gate. Installed outside the 0..gate_count loop (stubs 48-127
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
// the ring-3 exit path relies on.
const syscall_stub = @extern(*const anyopaque, .{ .name = "isr128" });
setGate(syscall_vector, @intFromPtr(syscall_stub));
idt[syscall_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
loadOnThisCpu();
}
@@ -145,9 +165,12 @@ pub fn loadOnThisCpu() void {
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
/// assembly stubs can `call` it by name. Exceptions are terminal; device
/// interrupts run their handler, get acknowledged, and return.
export fn interruptDispatch(state: *const CpuState) callconv(.c) void {
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
if (state.vector < 32) {
on_fault(state); // CPU exception — never returns
} else if (state.vector == syscall_vector) {
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
if (syscall_handler) |handler| handler(state);
} else if (handlers[state.vector]) |handler| {
// Acknowledge before running the handler: a handler that switches tasks
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
+94
View File
@@ -76,6 +76,96 @@ task_trampoline:
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# --- ring 3 entry/exit ------------------------------------------------------
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
# grow safely into it — then builds the 5-word iretq frame with the user
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
# timer keeps running in user mode.
.global enter_user
enter_user:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
iretq
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
# and stay off — the Zig caller re-enables.
.global user_exit_to_kernel
user_exit_to_kernel:
mov user_saved_rsp(%rip), %rsp
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret
.section .bss
.balign 8
user_saved_rsp:
.skip 8
.text
# --- user-mode test programs -------------------------------------------------
# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and
# entered via enter_user. Position-independent (immediates and short jumps only).
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
# from the user mapping.
.section .rodata
# The hello program: ping syscall (rax=1) with a value, a delay loop long enough
# for several timer ticks to land while at CPL 3 (proving interrupt-from-user +
# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety
# net in case exit ever returns.
.global user_prog_start
.global user_prog_end
user_prog_start:
mov $1, %rax
mov $0xC0DE, %rdi
int $0x80
mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG
1: dec %rcx
jnz 1b
mov $1, %rax
mov $0xBEEF, %rdi
int $0x80
mov $0, %rax
int $0x80
2: jmp 2b
user_prog_end:
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# page is unconditionally mapped supervisor (paging.zig), so this must take a
# #PF with error code 0x5 (present | user) before any access happens. The
# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax`
# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page.
.global user_pf_start
.global user_pf_end
user_pf_start:
mov $0xFEE00000, %ecx
mov (%rcx), %rax
1: jmp 1b
user_pf_end:
.text
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec
.global isr\vec
@@ -145,6 +235,10 @@ STUB_NOERR 45
STUB_NOERR 46
STUB_NOERR 47
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
# vector; dispatched specially in interruptDispatch.
STUB_NOERR 128
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
+29
View File
@@ -18,6 +18,7 @@ const page_size = danos.page_size;
// Page-table entry bits.
const present: u64 = 1 << 0;
const writable: u64 = 1 << 1;
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
const no_execute: u64 = 1 << 63;
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
@@ -128,6 +129,34 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void {
invalidate(virt);
}
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
/// no kernel mapping's protection is widened (the leaf still governs).
fn descendUser(entry: *u64) u64 {
const table = descend(entry);
entry.* |= user;
return table;
}
/// Map one 4 KiB page `virt` -> `phys` accessible from ring 3. W^X is the
/// caller's contract: code pages are read-only + executable, data pages are
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
/// `descendUser`).
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
var flags: u64 = present | user;
if (writable_page) flags |= writable;
if (!executable) flags |= no_execute;
const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags;
invalidate(virt);
}
/// Whether `virt` is currently mapped **executable** — present with the NX bit
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
/// case). Returns false if unmapped. Used for W^X checks in tests.
+18 -6
View File
@@ -1,9 +1,11 @@
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
//! switches to that stack when the exception fires — no matter how broken the
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
//! that happens *because* the current stack is unusable still lands on solid
//! ground instead of triple-faulting.
//! Task State Segment and its interrupt stacks. In long mode the TSS has two
//! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry,
//! and the CPU switches to that stack when the exception fires — no matter how
//! broken the interrupted stack was. We use IST1 for the double-fault handler,
//! so a fault that happens *because* the current stack is unusable still lands
//! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack
//! the CPU switches to when an interrupt arrives from ring 3 (published by the
//! user-mode entry path via `rsp0Ptr`).
//!
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
@@ -50,6 +52,16 @@ var ap_ist_top = [_]usize{0} ** max_cpus; // per-AP IST stack top (0 = BSP / not
/// Loads the task register with the TSS selector. Defined in isr.s.
extern fn load_tr(selector: u16) callconv(.c) void;
/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a
/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset,
/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a
/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path
/// (enter_user in isr.s) writes the current kernel stack pointer through this
/// before dropping to user mode.
pub fn rsp0Ptr(cpu: usize) *align(4) u64 {
return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4);
}
/// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the
/// BSP before waking that core; read by the core's own `setupThisCpu`.
pub fn setApIstStack(cpu: usize, top: usize) void {
+20 -1
View File
@@ -7,6 +7,7 @@ const log = @import("log.zig");
const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const scheduler = @import("scheduler.zig");
const usermode = @import("usermode.zig");
const platform = @import("platform");
const tests = @import("tests.zig");
const build_options = @import("build_options");
@@ -245,7 +246,25 @@ fn kmain(boot_info: *const BootInfo) noreturn {
log.checkpoint(cp_running);
status("kernel initialised.\n");
// TODO: init process
// Hand over to user space: run /sbin/init (read off the boot volume by the
// loader) in ring 3. Preemption is off for the run — the M1 user-mode path
// publishes *this* core's TSS.rsp0 and must not migrate (the flag is
// global, so the system goes cooperative meanwhile; the other cores are
// idle). init becomes a real schedulable process in M3.
if (boot_info.init_len != 0) {
status("starting /sbin/init...\n");
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
scheduler.setPreemption(false);
const code = usermode.runInitElf(image);
scheduler.setPreemption(true);
if (code) |c| {
statusPrint("/sbin/init exited with code {d}.\n", .{c});
} else |err| {
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
}
} else {
status("no /sbin/init on the boot volume.\n");
}
status("\nnothing left to do; halting CPU.\n");
+68
View File
@@ -17,6 +17,7 @@ const pmm = @import("pmm.zig");
const heap = @import("heap.zig");
const sched = @import("scheduler.zig");
const ipc = @import("ipc.zig");
const usermode = @import("usermode.zig");
/// Formatted write straight to serial, independent of the framebuffer console.
fn log(comptime fmt: []const u8, args: anytype) void {
@@ -92,6 +93,12 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
faultNoExecute();
} else if (eql(case, "fault-null")) {
faultNull();
} else if (eql(case, "user")) {
userTest();
} else if (eql(case, "user-pf")) {
userPfTest();
} else if (eql(case, "init")) {
initTest(boot_info);
} else if (eql(case, "poweroff")) {
powerTest(.off);
} else if (eql(case, "reboot")) {
@@ -702,6 +709,67 @@ fn smpRetryTest() void {
result();
}
// --- ring 3 (user mode) -----------------------------------------------------
/// The full ring-3 round trip: enter user mode, take syscalls and timer
/// interrupts from CPL 3, and come back. Preemption is disabled for the run —
/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate
/// (interrupts still fire and iretq back into ring 3, which is the point).
fn userTest() void {
log("DANOS-TEST-BEGIN: user\n", .{});
sched.setPreemption(false);
const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: {
log("DANOS-USER: run failed: {s}\n", .{@errorName(err)});
break :blk false;
};
sched.setPreemption(true);
check("user program ran and exited (ring-3 round trip)", ran);
check("two ping syscalls received", usermode.ping_count == 2);
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23);
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
result();
}
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
/// RIP. The fault report is the pass signal (matched by the harness); if the
/// read is somehow allowed the blob spins and the harness times out.
fn userPfTest() void {
log("DANOS-TEST-BEGIN: user-pf\n", .{});
sched.setPreemption(false);
_ = usermode.run(usermode.pfBlob()) catch {};
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
}
/// The full user-binary path: the bootloader read sbin/init off the boot
/// volume and handed it over; load it as a user ELF and run it in ring 3. The
/// same call the normal boot path makes — here with teeth.
fn initTest(boot_info: *const BootInfo) void {
log("DANOS-TEST-BEGIN: init\n", .{});
check("bootloader handed over sbin/init", boot_info.init_len != 0);
if (boot_info.init_len == 0) {
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
sched.setPreemption(false); // see userTest: pins the run to this core's rsp0
const code = usermode.runInitElf(image);
sched.setPreemption(true);
if (code) |c| {
check("init loaded, ran, and exited (user ELF path)", true);
check("init exited cleanly (code 0)", c == 0);
} else |err| {
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
check("init loaded, ran, and exited (user ELF path)", false);
}
check("init's write arrived intact", eql(usermode.write_buf[0..usermode.write_len], "init: hello from user space\n"));
check("write came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23);
result();
}
fn faultInvalidOpcode() void {
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
asm volatile ("ud2");
+298
View File
@@ -0,0 +1,298 @@
//! Ring-3 execution — the isolation track's first rung (M1). Two entry points:
//! `run` executes a raw code blob (the user/user-pf test programs), and
//! `runInitElf` loads and runs a real user ELF (`/sbin/init`, handed over by
//! the bootloader). Both run at CPL 3 in the current (global) page tables:
//! frames are mapped user-accessible with W^X (code RO+X, data RW+NX), the CPU
//! drops privilege via iretq, and the program talks to the kernel only through
//! the `int $0x80` syscall gate. No processes or per-task address spaces yet —
//! this proves the privilege mechanisms (user descriptors, TSS.rsp0, the U/S
//! page bit, ring transitions) that M3 builds on.
//!
//! Caller contract (M1 limitations):
//! - enter/exit publishes TSS.rsp0 on the *current* core only, so the caller
//! must prevent migration for the duration — disable preemption around the
//! call (rsp0-per-context-switch arrives with real user tasks in M3).
//! - The kernel-side saved context (`user_saved_rsp` in isr.s) is a single
//! global: at most one core may be inside user mode at a time.
const std = @import("std");
const elf = std.elf;
const danos = @import("danos");
const arch = @import("arch");
const pmm = @import("pmm.zig");
const sched = @import("scheduler.zig");
const log = @import("log.zig");
const page_size = danos.page_size;
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
/// the identity map (low indices) and the vmm test address (index 128), so
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
/// An ELF image may occupy [code_virt, stack_virt); the stack page sits above.
pub const code_virt: u64 = 0x0000_7000_0000_0000;
pub const stack_virt: u64 = 0x0000_7000_0020_0000;
// The hand-assembled user program blobs (isr.s, .rodata).
const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" });
const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" });
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit.
pub fn helloBlob() []const u8 {
return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)];
}
/// The isolation-proof program: reads a kernel-only page, must #PF.
pub fn pfBlob() []const u8 {
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
}
/// What a ping syscall recorded — evidence for the test to assert on.
pub const Ping = struct { value: u64 = 0, cs: u64 = 0, ticks: u64 = 0 };
pub var pings = [_]Ping{.{}} ** 2;
pub var ping_count: usize = 0;
/// What write syscalls produced (accumulated), and the exit syscall's code.
pub var write_buf: [256]u8 = undefined;
pub var write_len: usize = 0;
pub var write_cs: u64 = 0;
pub var exit_code: u64 = 0;
/// The M1 syscall surface, dispatched on the saved user rax:
/// 0 = exit(code) — unwind back into the kernel context that entered
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
/// replaces this in M3; rax is written back as the return value already, since
/// isr_common restores user registers from the trap frame.
fn syscall(state: *arch.CpuState) void {
switch (state.rax) {
0 => {
exit_code = state.rdi;
arch.userExit();
},
1 => {
if (ping_count < pings.len) {
pings[ping_count] = .{ .value = state.rdi, .cs = state.cs, .ticks = arch.ticks() };
ping_count += 1;
}
state.rax = 0;
},
2 => {
// The pointer must lie inside the user region (image + stack) —
// kernel addresses and non-canonical values fall outside it, so the
// kernel-side read below can't be steered at kernel data. Length is
// checked first so the upper-bound subtraction can't underflow.
// Known gap (fine for a trusted init): a pointer into an *unmapped*
// hole in the region passes the check and the read #PFs -> on_fault
// halts — a self-DoS, not an isolation break. Copy-in with fault
// recovery is M3+ (with SMAP, once there's a reason to enable it).
const ptr = state.rdi;
const len = state.rsi;
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
const src: [*]const u8 = @ptrFromInt(ptr);
const n = @min(len, write_buf.len - write_len);
@memcpy(write_buf[write_len..][0..n], src[0..n]);
write_len += n;
write_cs = state.cs;
log.write("DANOS-INIT: ");
log.write(src[0..len]);
state.rax = len;
} else {
state.rax = @bitCast(@as(i64, -1));
}
},
else => state.rax = @bitCast(@as(i64, -1)),
}
}
/// Reset the recorded syscall evidence before a user-mode run.
fn resetRecords() void {
ping_count = 0;
write_len = 0;
write_cs = 0;
exit_code = 0;
}
pub const RunError = error{ ProgramTooBig, OutOfMemory };
/// Map `blob` at code_virt with a fresh user stack, drop to ring 3, and return
/// once the program exits via syscall 0. See the migration caveat in the module
/// doc. A program that faults instead never returns (on_fault halts the core).
pub fn run(blob: []const u8) RunError!void {
if (blob.len > page_size) return error.ProgramTooBig;
const code_frame = pmm.alloc() orelse return error.OutOfMemory;
const stack_frame = pmm.alloc() orelse {
pmm.free(code_frame);
return error.OutOfMemory;
};
// Fill the code frame through its identity mapping (supervisor RW): the
// user-facing mapping is read-only, and this also sidesteps CR0.WP/SMAP.
// The tail is padded with int3 so a stray jump traps instead of sliding.
const code: [*]u8 = @ptrFromInt(code_frame);
@memcpy(code[0..blob.len], blob);
@memset(code[blob.len..page_size], 0xCC);
arch.setSyscallHandler(syscall);
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
resetRecords();
arch.enterUser(sched.currentCpuIndex(), code_virt, stack_virt + page_size);
// Back via the exit syscall; the interrupt gate left IF clear.
arch.enableInterrupts();
arch.unmapPage(code_virt);
arch.unmapPage(stack_virt);
pmm.free(code_frame);
pmm.free(stack_frame);
}
// --- user ELF loading (/sbin/init) ------------------------------------------
pub const InitError = error{
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
BadEntry, // e_entry not inside an executable segment
ProgramTooBig, // more pages than the loader's budget
OutOfMemory,
};
const max_segments = 16;
const max_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
const Segment = struct {
vaddr: u64,
memsz: u64,
filesz: u64,
off: u64,
writable: bool,
executable: bool,
fn pages(self: Segment) u64 {
return (self.memsz + page_size - 1) / page_size;
}
};
/// Pages mapped for the running init image, for rollback if loading fails
/// midway. On success the mappings stay for the system's life — there's no
/// address-space teardown until real processes exist (M3).
var loaded = [_]struct { virt: u64, frame: u64 }{.{ .virt = 0, .frame = 0 }} ** (max_pages + 1);
var loaded_count: usize = 0;
/// Parse and validate every PT_LOAD before touching memory. Bounds are checked
/// against the image and the user region; segments must be page-aligned,
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
/// segment, mapped RO+NX).
fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!struct { count: usize, entry: u64 } {
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf;
if (ehdr.e_machine != .X86_64) return error.BadElf;
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
if (ehdr.e_phnum > max_segments) return error.BadElf;
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
var count: usize = 0;
var total_pages: u64 = 0;
for (0..ehdr.e_phnum) |i| {
const off = ehdr.e_phoff + i * ehdr.e_phentsize;
const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]);
if (phdr.p_type != elf.PT_LOAD) continue;
if (phdr.p_memsz == 0) continue;
if (phdr.p_vaddr % page_size != 0) return error.BadSegment;
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
// Inside the user image region, strictly below the stack page.
if (phdr.p_vaddr < code_virt) return error.BadSegment;
if (phdr.p_memsz > stack_virt - phdr.p_vaddr) return error.BadSegment;
const w = phdr.p_flags & elf.PF_W != 0;
const x = phdr.p_flags & elf.PF_X != 0;
if (w and x) return error.BadSegment; // W^X, even for init
const seg = Segment{
.vaddr = phdr.p_vaddr,
.memsz = phdr.p_memsz,
.filesz = phdr.p_filesz,
.off = phdr.p_offset,
.writable = w,
.executable = x,
};
// No overlap with any earlier segment (page-granular, since mapping is).
for (segs[0..count]) |other| {
const a_end = seg.vaddr + seg.pages() * page_size;
const b_end = other.vaddr + other.pages() * page_size;
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
}
total_pages += seg.pages();
if (total_pages > max_pages) return error.ProgramTooBig;
segs[count] = seg;
count += 1;
}
if (count == 0) return error.BadElf;
// The entry point must land inside an executable segment.
for (segs[0..count]) |seg| {
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
return .{ .count = count, .entry = ehdr.e_entry };
}
return error.BadEntry;
}
/// Map one page of a segment: fresh frame, zero + copy through the identity
/// mapping (the user mapping may be read-only), then the user-visible mapping
/// with the segment's W^X permissions. Records the page for rollback.
fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void {
const frame = pmm.alloc() orelse return error.OutOfMemory;
const dst: [*]u8 = @ptrFromInt(frame);
@memset(dst[0..page_size], 0);
const page_off = page_index * page_size;
if (page_off < seg.filesz) {
const n = @min(page_size, seg.filesz - page_off);
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
}
const virt = seg.vaddr + page_off;
arch.mapUserPage(virt, frame, seg.writable, seg.executable);
loaded[loaded_count] = .{ .virt = virt, .frame = frame };
loaded_count += 1;
}
fn unloadAll() void {
for (loaded[0..loaded_count]) |p| {
arch.unmapPage(p.virt);
pmm.free(p.frame);
}
loaded_count = 0;
}
/// Load a user ELF image, run it in ring 3 from its entry point, and return its
/// exit code. Same caller contract as `run` (preemption off, one core). On
/// success the user mappings are left in place — teardown comes with real
/// processes (M3); on a loading error everything is rolled back.
pub fn runInitElf(image: []const u8) InitError!u64 {
var segs: [max_segments]Segment = undefined;
const parsed = try parseSegments(image, &segs);
loaded_count = 0;
errdefer unloadAll();
for (segs[0..parsed.count]) |seg| {
for (0..seg.pages()) |i| try loadPage(image, seg, i);
}
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame };
loaded_count += 1;
arch.setSyscallHandler(syscall);
resetRecords();
arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size);
arch.enableInterrupts(); // the exit arrived through an interrupt gate
return exit_code;
}
+6
View File
@@ -109,4 +109,10 @@ pub const BootInfo = extern struct {
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
/// field instead, so the kernel discovers devices without knowing what booted it.
acpi_rsdp: u64 = 0,
/// The raw `/sbin/init` ELF image, read off the boot volume by the loader
/// into memory that survives the handoff (classified reserved, so the kernel
/// identity-maps it and never allocates over it). 0/0 = no init found — the
/// kernel boots without user space. Grows into a full initrd handoff later.
init_base: u64 = 0,
init_len: u64 = 0,
};