isolation M1: ring 3 + a real /sbin/init, end to end

Ring 3 works: user GDT descriptors (sysret-ready layout), TSS.rsp0,
U/S-bit user mappings (W^X preserved), an int 0x80 syscall gate with a
mutable trap frame, and a setjmp-style enter/exit path. /sbin/init is a
real freestanding Zig binary built from sbin/, shipped on the ESP,
loaded by the bootloader (BootInfo.init_base/len), validated and mapped
by an in-kernel user-ELF loader, and run at CPL 3 — syscalls: exit,
ping, write. Tests: user, user-pf (U/S isolation proof, error code
0x5), init. Suite 27/27.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-08 22:15:07 +01:00
co-authored by Claude Fable 5
parent 7501bd1703
commit 546dd44a2a
17 changed files with 838 additions and 25 deletions
+30
View File
@@ -145,6 +145,31 @@ pub fn build(b: *std.Build) void {
b.installArtifact(exe); b.installArtifact(exe);
// --- /sbin/init: the first user-space program ---
// Its own tiny freestanding binary, linked at a fixed address inside the
// kernel's user region (usermode.zig) and started in ring 3 by the kernel's
// user-ELF loader. `.large` because the image base is above 4 GiB — small/
// medium code models emit 32-bit absolute relocations that can't reach.
// Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its
// size has no reason to track the kernel's optimize mode.
const init_exe = b.addExecutable(.{
.name = "init",
.root_module = b.createModule(.{
.root_source_file = b.path("sbin/init.zig"),
.target = kernel_target,
.optimize = .ReleaseSmall,
.code_model = .large,
.single_threaded = true,
.sanitize_c = .off,
.stack_check = false,
.stack_protector = false,
}),
});
init_exe.setLinkerScript(b.path("sbin/linker.ld"));
init_exe.entry = .{ .symbol_name = "_start" };
init_exe.image_base = 0x7000_0000_0000;
b.installArtifact(init_exe);
// Boot methods live in src/boot/, one per way of getting the kernel running. // Boot methods live in src/boot/, one per way of getting the kernel running.
// Each is its own binary/entry (a loader is built for its own target); today // Each is its own binary/entry (a loader is built for its own target); today
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis. // that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
@@ -203,6 +228,10 @@ pub fn build(b: *std.Build) void {
const kernel_install = b.addInstallArtifact(exe, .{ const kernel_install = b.addInstallArtifact(exe, .{
.dest_dir = .{ .override = .{ .custom = "esp" } }, .dest_dir = .{ .override = .{ .custom = "esp" } },
}); });
// The bootloader loads init from sbin/init on the same volume.
const init_install = b.addInstallArtifact(init_exe, .{
.dest_dir = .{ .override = .{ .custom = "esp/sbin" } },
});
// The firmware needs to write NVRAM, so give it a writable copy of the vars. // The firmware needs to write NVRAM, so give it a writable copy of the vars.
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars }); const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
@@ -239,6 +268,7 @@ pub fn build(b: *std.Build) void {
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) }); run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
run_efi.step.dependOn(&efi_install.step); run_efi.step.dependOn(&efi_install.step);
run_efi.step.dependOn(&kernel_install.step); run_efi.step.dependOn(&kernel_install.step);
run_efi.step.dependOn(&init_install.step);
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log"); const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
run_efi_step.dependOn(&run_efi.step); run_efi_step.dependOn(&run_efi.step);
+7 -1
View File
@@ -1,5 +1,11 @@
# System Calls # System Calls
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel). System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
> **Status:** danos currently has a placeholder M1 surface behind an `int 0x80`
> gate — `0 = exit(code)`, `1 = ping(value)`, `2 = write(ptr, len)` (see
> `src/kernel/usermode.zig`, used by `sbin/init.zig`). It exists to prove the
> ring transition; the microkernel set below replaces it (via `syscall`/`sysret`)
> when processes land (M3).
## The Mechanism of a Syscall ## The Mechanism of a Syscall
+8 -3
View File
@@ -81,11 +81,16 @@ prerequisites.
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and (with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking, [heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md). in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), and
**ring 3**: user GDT/TSS plumbing, U/S-bit mappings, an `int 0x80` syscall gate, and
`/sbin/init` — a real user ELF built from `sbin/`, shipped on the boot volume, loaded
by the kernel, run at CPL 3 — plus a [test harness](testing.md).
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel, - **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
ring 3, per-process page tables). The substrate everything else needs. *Next, and a ring 3, per-process page tables). The substrate everything else needs. *In
prerequisite for the resilience and driver tracks.* progress: ring 3 + a loaded `/sbin/init` work (M1); next the higher-half move (M2),
then per-process address spaces + `syscall`/`sysret` + init as a real schedulable
process (M3), then ELF/initrd generalisation (M4).*
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server, - **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
resource cleanup on death, then a restartable driver as proof. Needs isolation. resource cleanup on death, then a restartable driver as proof. Needs isolation.
See [resilience.md](resilience.md). See [resilience.md](resilience.md).
+54
View File
@@ -0,0 +1,54 @@
//! /sbin/init — the first user-space program. Built as its own freestanding
//! binary (see build.zig), shipped on the boot volume at sbin/init, loaded by
//! the bootloader, and started in ring 3 by the kernel's user-ELF loader
//! (src/kernel/usermode.zig). It talks to the kernel only through the
//! `int $0x80` syscall gate.
//!
//! Today it just proves the path — say hello, exit — and grows into the real
//! init (service supervision) once processes are schedulable (M3).
const std = @import("std");
// The M1 syscall numbers (usermode.zig): 0 = exit(code), 2 = write(ptr, len).
const sys_exit = 0;
const sys_write = 2;
fn syscall2(n: u64, a: u64, b: u64) u64 {
return asm volatile ("int $0x80"
: [ret] "={rax}" (-> u64),
: [n] "{rax}" (n),
[a] "{rdi}" (a),
[b] "{rsi}" (b),
: .{ .memory = true });
}
fn write(msg: []const u8) void {
_ = syscall2(sys_write, @intFromPtr(msg.ptr), msg.len);
}
fn exit(code: u64) noreturn {
_ = syscall2(sys_exit, code, 0);
unreachable; // the kernel never returns from exit
}
/// Entry. Naked: the kernel enters with rsp 16-aligned, but a SysV function
/// expects rsp ≡ 8 (mod 16) on entry (as if reached by `call`) — so re-enter
/// the ABI with an actual call. The trap after is unreachable.
pub export fn _start() callconv(.naked) noreturn {
asm volatile (
\\call init_main
\\ud2
);
}
export fn init_main() callconv(.c) noreturn {
write("init: hello from user space\n");
exit(0);
}
/// No runtime to unwind into — report the panic as a nonzero exit code.
pub const panic = std.debug.FullPanic(struct {
fn panic(_: []const u8, _: ?usize) noreturn {
exit(127);
}
}.panic);
+45
View File
@@ -0,0 +1,45 @@
/* /sbin/init link layout.
*
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
* inside the kernel's user region). Same discipline as the kernel's script:
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
* user-ELF loader can map each segment with exact W^X permissions. Note the
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
* base, so the entry point comes from e_entry, not the base address.
*/
ENTRY(_start)
/* FLAGS bits: 1=X, 2=W, 4=R. */
PHDRS {
text PT_LOAD FLAGS(5); /* R + X */
rodata PT_LOAD FLAGS(4); /* R */
data PT_LOAD FLAGS(6); /* R + W */
}
SECTIONS {
.text ALIGN(4K) : {
*(.text .text.*)
} :text
.rodata ALIGN(4K) : {
*(.rodata .rodata.*)
} :rodata
.data ALIGN(4K) : {
*(.data .data.*)
} :data
/* .bss occupies memory but not file space; the loader zeroes the
* filesz..memsz gap. */
.bss ALIGN(4K) : {
*(.bss .bss.*)
*(COMMON)
} :data
/DISCARD/ : {
*(.comment)
*(.note .note.*)
*(.eh_frame .eh_frame_hdr)
}
}
+48
View File
@@ -11,6 +11,10 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
/// build.zig). UEFI wants a UTF-16, null-terminated path. /// build.zig). UEFI wants a UTF-16, null-terminated path.
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel"); const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
/// Path of the init program on the boot volume (UEFI paths use backslashes;
/// the FAT driver walks the components itself, so no directory dance needed).
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("sbin\\init");
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file. /// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
const page_size = 4096; const page_size = 4096;
const seek_end = 0xffff_ffff_ffff_ffff; const seek_end = 0xffff_ffff_ffff_ffff;
@@ -54,6 +58,13 @@ fn boot() !noreturn {
const entry = try loadKernel(bs, &boot_info); const entry = try loadKernel(bs, &boot_info);
// Best effort: a volume without sbin/init still boots (kernel-only).
loadInit(bs, &boot_info) catch |err| {
log("danos: no sbin/init (");
logBytes(@errorName(err));
log(") - booting without user space\r\n");
};
log("danos: kernel loaded, exiting boot services\r\n"); log("danos: kernel loaded, exiting boot services\r\n");
boot_info.memory_map = try exitBootServices(bs); boot_info.memory_map = try exitBootServices(bs);
@@ -198,6 +209,43 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
return loadElf(bs, image, boot_info); return loadElf(bs, image, boot_info);
} }
/// Read the init program (sbin/init) into memory that outlives the loader and
/// record it in the handoff. The pool buffer is deliberately NOT freed: it's
/// LoaderData, which the memory-map conversion classifies as reserved, so the
/// kernel identity-maps it and reads the ELF from there. The kernel does the
/// loading itself (into ring-3 mappings) — the loader just ferries the bytes.
fn loadInit(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void {
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
return error.NoLoadedImage;
const device = loaded.device_handle orelse return error.NoBootDevice;
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
return error.NoFileSystem;
const root = try fs.openVolume();
defer _ = root.close() catch {};
const file = try root.open(init_file_name, .read, .{});
defer _ = file.close() catch {};
try file.setPosition(seek_end);
const size: usize = @intCast(try file.getPosition());
try file.setPosition(0);
if (size == 0) return error.EmptyFile;
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
var read_total: usize = 0;
while (read_total < size) {
const n = try file.read(image[read_total..]);
if (n == 0) return error.UnexpectedEof;
read_total += n;
}
boot_info.init_base = @intFromPtr(image.ptr);
boot_info.init_len = size;
log("danos: sbin/init loaded\r\n");
}
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and /// Validate the ELF, copy every PT_LOAD segment to its physical address, and
/// record each segment's layout so the kernel can re-map itself with the right /// record each segment's layout so the kernel can re-map itself with the right
/// permissions. /// permissions.
+44 -3
View File
@@ -76,6 +76,45 @@ pub fn unmapPage(virt: u64) void {
paging.unmap(virt); paging.unmap(virt);
} }
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
/// W^X: code read-only + executable, data writable + no-execute.
pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void {
paging.mapUser(virt, phys, writable, executable);
}
// --- ring 3 entry/exit -----------------------------------------------------
/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context,
/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0,
/// so ring-3 interrupts land on a good stack), builds an iretq frame with the
/// user selectors, and iretq's. "Returns" only when the user program triggers
/// the exit path (user_exit_to_kernel).
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
/// `enter_user` had returned (defined in isr.s). Called by the exit syscall.
extern fn user_exit_to_kernel() callconv(.c) noreturn;
/// Run user code at `rip` with stack `rsp` on this core (`cpu` = the caller's CPU
/// index; the arch layer can't ask the scheduler). Returns after the user program
/// exits via syscall. Interrupts are disabled on return (the exit arrives through
/// an interrupt gate) — the caller re-enables.
pub fn enterUser(cpu: usize, rip: u64, rsp: u64) void {
enter_user(rip, rsp, tss.rsp0Ptr(cpu));
}
/// Never returns to the user program: unwind to the kernel context that called
/// `enterUser`. For the exit syscall's handler.
pub fn userExit() noreturn {
user_exit_to_kernel();
}
/// Register the handler for the ring-3 syscall gate (int 0x80, vector 128). The
/// handler may write the trap frame (e.g. rax as the return value).
pub fn setSyscallHandler(handler: *const fn (*idt.CpuState) void) void {
idt.setSyscallHandler(handler);
}
/// CR3 holds the physical address of the active top-level page table. /// CR3 holds the physical address of the active top-level page table.
pub fn readCr3() u64 { pub fn readCr3() u64 {
return asm volatile ("mov %%cr3, %[out]" return asm volatile ("mov %%cr3, %[out]"
@@ -84,9 +123,11 @@ pub fn readCr3() u64 {
} }
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU /// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
/// data pointer (there's no user mode yet, so no `swapgs` dance — GS base is always /// data pointer. Because it's always read back via `rdmsr` (never gs-relative
/// the running core's per-CPU block). Set once per core during bring-up, after the /// addressing), no `swapgs` dance is needed even with user mode: the MSR is
/// GDT is loaded (loading a GS *selector* would otherwise clobber this base). /// privileged, ring 3 can't touch it, and its value is unaffected by ring
/// transitions. Set once per core during bring-up, after the GDT is loaded
/// (loading a GS *selector* would otherwise clobber this base).
const ia32_gs_base = 0xC000_0101; const ia32_gs_base = 0xC000_0101;
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each /// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
+25 -10
View File
@@ -1,8 +1,8 @@
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but //! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates //! the CPU still needs valid code/data segment descriptors, and the IDT's gates
//! reference a code selector — so we install our own flat GDT with known //! reference a code selector — so we install our own flat GDT with known
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever //! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3)
//! the firmware left in place. //! rather than trusting whatever the firmware left in place.
//! //!
//! The code/data descriptors are identical on every core, but the **TSS descriptor //! The code/data descriptors are identical on every core, but the **TSS descriptor
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see //! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
@@ -14,20 +14,35 @@ const config = @import("config");
/// Selectors into the table (index * 8). Same on every core's GDT. /// Selectors into the table (index * 8). Same on every core's GDT.
pub const kernel_code = 0x08; pub const kernel_code = 0x08;
pub const kernel_data = 0x10; pub const kernel_data = 0x10;
pub const tss_selector = 0x18; pub const user_data = 0x18;
pub const user_code = 0x20;
pub const tss_selector = 0x28;
/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data*
/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS
/// raises #GP(0).
pub const user_code_rpl3 = user_code | 3;
pub const user_data_rpl3 = user_data | 3;
const max_cpus = config.max_cpus; const max_cpus = config.max_cpus;
const entries = 5; // null, code, data, TSS-low, TSS-high const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor, /// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
/// filled in per core by `setTssFor`. /// filled in per core by `setTssFor`.
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF /// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF /// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF
/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF
/// User data sits below user code so a future SYSRET works unchanged: it loads
/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10
/// those land on 0x20 (user code) and 0x18 (user data).
const template = [entries]u64{ const template = [entries]u64{
0, // null descriptor (required) 0, // null descriptor (required)
0x00AF9A000000FFFF, // kernel code (0x08) 0x00AF9A000000FFFF, // kernel code (0x08)
0x00CF92000000FFFF, // kernel data (0x10) 0x00CF92000000FFFF, // kernel data (0x10)
0, // TSS descriptor low (0x18) 0x00CFF2000000FFFF, // user data (0x18)
0x00AFFA000000FFFF, // user code (0x20)
0, // TSS descriptor low (0x28)
0, // TSS descriptor high 0, // TSS descriptor high
}; };
@@ -38,13 +53,13 @@ var gdts = [_][entries]u64{template} ** max_cpus;
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit /// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
/// TSS. Write it into that core's GDT before it loads the TSS selector. /// TSS. Write it into that core's GDT before it loads the TSS selector.
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void { pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
gdts[cpu][3] = (limit & 0xFFFF) | gdts[cpu][5] = (limit & 0xFFFF) |
((base & 0xFFFF) << 16) | ((base & 0xFFFF) << 16) |
(((base >> 16) & 0xFF) << 32) | (((base >> 16) & 0xFF) << 32) |
(@as(u64, 0x89) << 40) | (@as(u64, 0x89) << 40) |
(((limit >> 16) & 0xF) << 48) | (((limit >> 16) & 0xF) << 48) |
(((base >> 24) & 0xFF) << 56); (((base >> 24) & 0xFF) << 56);
gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF; gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF;
} }
/// The operand `lgdt` wants: table byte-length minus one, then its address. /// The operand `lgdt` wants: table byte-length minus one, then its address.
+24 -1
View File
@@ -26,6 +26,19 @@ pub fn setHandler(vector: usize, handler: Handler) void {
handlers[vector] = handler; handlers[vector] = handler;
} }
/// The ring-3 syscall gate's vector (`int $0x80`, the classic choice — well away
/// from the device range) and its handler. Unlike device handlers, a syscall
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
/// user registers and writes rax as the return value, which isr_common then
/// restores into the user context.
pub const syscall_vector = 128;
var syscall_handler: ?*const fn (*CpuState) void = null;
pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void {
syscall_handler = handler;
}
/// The register + trap frame the ISR stubs build on the stack, laid out so the /// The register + trap frame the ISR stubs build on the stack, laid out so the
/// lowest address (where RSP points when we call the handler) is the first field. /// lowest address (where RSP points when we call the handler) is the first field.
/// See the push order in `isrCommon` below. /// See the push order in `isrCommon` below.
@@ -128,6 +141,13 @@ pub fn init() void {
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the // Run the double-fault handler (vector 8) on IST1: a #DF usually means the
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig. // current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
idt[8].ist = tss.double_fault_ist; idt[8].ist = tss.double_fault_ist;
// The syscall gate. Installed outside the 0..gate_count loop (stubs 48-127
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
// the ring-3 exit path relies on.
const syscall_stub = @extern(*const anyopaque, .{ .name = "isr128" });
setGate(syscall_vector, @intFromPtr(syscall_stub));
idt[syscall_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
loadOnThisCpu(); loadOnThisCpu();
} }
@@ -145,9 +165,12 @@ pub fn loadOnThisCpu() void {
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the /// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
/// assembly stubs can `call` it by name. Exceptions are terminal; device /// assembly stubs can `call` it by name. Exceptions are terminal; device
/// interrupts run their handler, get acknowledged, and return. /// interrupts run their handler, get acknowledged, and return.
export fn interruptDispatch(state: *const CpuState) callconv(.c) void { export fn interruptDispatch(state: *CpuState) callconv(.c) void {
if (state.vector < 32) { if (state.vector < 32) {
on_fault(state); // CPU exception — never returns on_fault(state); // CPU exception — never returns
} else if (state.vector == syscall_vector) {
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
if (syscall_handler) |handler| handler(state);
} else if (handlers[state.vector]) |handler| { } else if (handlers[state.vector]) |handler| {
// Acknowledge before running the handler: a handler that switches tasks // Acknowledge before running the handler: a handler that switches tasks
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on // (the scheduler) may not return promptly, and the LAPIC mustn't wait on
+94
View File
@@ -76,6 +76,96 @@ task_trampoline:
1: hlt # if the entry returns, idle (still preemptible) 1: hlt # if the entry returns, idle (still preemptible)
jmp 1b jmp 1b
# --- ring 3 entry/exit ------------------------------------------------------
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
# grow safely into it — then builds the 5-word iretq frame with the user
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
# timer keeps running in user mode.
.global enter_user
enter_user:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
iretq
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
# and stay off — the Zig caller re-enables.
.global user_exit_to_kernel
user_exit_to_kernel:
mov user_saved_rsp(%rip), %rsp
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret
.section .bss
.balign 8
user_saved_rsp:
.skip 8
.text
# --- user-mode test programs -------------------------------------------------
# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and
# entered via enter_user. Position-independent (immediates and short jumps only).
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
# from the user mapping.
.section .rodata
# The hello program: ping syscall (rax=1) with a value, a delay loop long enough
# for several timer ticks to land while at CPL 3 (proving interrupt-from-user +
# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety
# net in case exit ever returns.
.global user_prog_start
.global user_prog_end
user_prog_start:
mov $1, %rax
mov $0xC0DE, %rdi
int $0x80
mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG
1: dec %rcx
jnz 1b
mov $1, %rax
mov $0xBEEF, %rdi
int $0x80
mov $0, %rax
int $0x80
2: jmp 2b
user_prog_end:
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# page is unconditionally mapped supervisor (paging.zig), so this must take a
# #PF with error code 0x5 (present | user) before any access happens. The
# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax`
# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page.
.global user_pf_start
.global user_pf_end
user_pf_start:
mov $0xFEE00000, %ecx
mov (%rcx), %rax
1: jmp 1b
user_pf_end:
.text
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0. # Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec .macro STUB_NOERR vec
.global isr\vec .global isr\vec
@@ -145,6 +235,10 @@ STUB_NOERR 45
STUB_NOERR 46 STUB_NOERR 46
STUB_NOERR 47 STUB_NOERR 47
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
# vector; dispatched specially in interruptDispatch.
STUB_NOERR 128
.extern interruptDispatch .extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order. # Shared tail. Register push order here defines the CpuState field order.
+29
View File
@@ -18,6 +18,7 @@ const page_size = danos.page_size;
// Page-table entry bits. // Page-table entry bits.
const present: u64 = 1 << 0; const present: u64 = 1 << 0;
const writable: u64 = 1 << 1; const writable: u64 = 1 << 1;
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
const no_execute: u64 = 1 << 63; const no_execute: u64 = 1 << 63;
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
@@ -128,6 +129,34 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void {
invalidate(virt); invalidate(virt);
} }
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
/// no kernel mapping's protection is widened (the leaf still governs).
fn descendUser(entry: *u64) u64 {
const table = descend(entry);
entry.* |= user;
return table;
}
/// Map one 4 KiB page `virt` -> `phys` accessible from ring 3. W^X is the
/// caller's contract: code pages are read-only + executable, data pages are
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
/// `descendUser`).
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
var flags: u64 = present | user;
if (writable_page) flags |= writable;
if (!executable) flags |= no_execute;
const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags;
invalidate(virt);
}
/// Whether `virt` is currently mapped **executable** — present with the NX bit /// Whether `virt` is currently mapped **executable** — present with the NX bit
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page /// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
/// case). Returns false if unmapped. Used for W^X checks in tests. /// case). Returns false if unmapped. Used for W^X checks in tests.
+18 -6
View File
@@ -1,9 +1,11 @@
//! Task State Segment and its interrupt stack. In long mode the TSS's main job //! Task State Segment and its interrupt stacks. In long mode the TSS has two
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU //! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry,
//! switches to that stack when the exception fires — no matter how broken the //! and the CPU switches to that stack when the exception fires — no matter how
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault //! broken the interrupted stack was. We use IST1 for the double-fault handler,
//! that happens *because* the current stack is unusable still lands on solid //! so a fault that happens *because* the current stack is unusable still lands
//! ground instead of triple-faulting. //! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack
//! the CPU switches to when an interrupt arrives from ring 3 (published by the
//! user-mode entry path via `rsp0Ptr`).
//! //!
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at //! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
//! once can't share one fault stack. So the TSS and its IST stack are per-core, //! once can't share one fault stack. So the TSS and its IST stack are per-core,
@@ -50,6 +52,16 @@ var ap_ist_top = [_]usize{0} ** max_cpus; // per-AP IST stack top (0 = BSP / not
/// Loads the task register with the TSS selector. Defined in isr.s. /// Loads the task register with the TSS selector. Defined in isr.s.
extern fn load_tr(selector: u16) callconv(.c) void; extern fn load_tr(selector: u16) callconv(.c) void;
/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a
/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset,
/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a
/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path
/// (enter_user in isr.s) writes the current kernel stack pointer through this
/// before dropping to user mode.
pub fn rsp0Ptr(cpu: usize) *align(4) u64 {
return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4);
}
/// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the /// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the
/// BSP before waking that core; read by the core's own `setupThisCpu`. /// BSP before waking that core; read by the core's own `setupThisCpu`.
pub fn setApIstStack(cpu: usize, top: usize) void { pub fn setApIstStack(cpu: usize, top: usize) void {
+20 -1
View File
@@ -7,6 +7,7 @@ const log = @import("log.zig");
const pmm = @import("pmm.zig"); const pmm = @import("pmm.zig");
const heap = @import("heap.zig"); const heap = @import("heap.zig");
const scheduler = @import("scheduler.zig"); const scheduler = @import("scheduler.zig");
const usermode = @import("usermode.zig");
const platform = @import("platform"); const platform = @import("platform");
const tests = @import("tests.zig"); const tests = @import("tests.zig");
const build_options = @import("build_options"); const build_options = @import("build_options");
@@ -245,7 +246,25 @@ fn kmain(boot_info: *const BootInfo) noreturn {
log.checkpoint(cp_running); log.checkpoint(cp_running);
status("kernel initialised.\n"); status("kernel initialised.\n");
// TODO: init process // Hand over to user space: run /sbin/init (read off the boot volume by the
// loader) in ring 3. Preemption is off for the run — the M1 user-mode path
// publishes *this* core's TSS.rsp0 and must not migrate (the flag is
// global, so the system goes cooperative meanwhile; the other cores are
// idle). init becomes a real schedulable process in M3.
if (boot_info.init_len != 0) {
status("starting /sbin/init...\n");
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
scheduler.setPreemption(false);
const code = usermode.runInitElf(image);
scheduler.setPreemption(true);
if (code) |c| {
statusPrint("/sbin/init exited with code {d}.\n", .{c});
} else |err| {
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
}
} else {
status("no /sbin/init on the boot volume.\n");
}
status("\nnothing left to do; halting CPU.\n"); status("\nnothing left to do; halting CPU.\n");
+68
View File
@@ -17,6 +17,7 @@ const pmm = @import("pmm.zig");
const heap = @import("heap.zig"); const heap = @import("heap.zig");
const sched = @import("scheduler.zig"); const sched = @import("scheduler.zig");
const ipc = @import("ipc.zig"); const ipc = @import("ipc.zig");
const usermode = @import("usermode.zig");
/// Formatted write straight to serial, independent of the framebuffer console. /// Formatted write straight to serial, independent of the framebuffer console.
fn log(comptime fmt: []const u8, args: anytype) void { fn log(comptime fmt: []const u8, args: anytype) void {
@@ -92,6 +93,12 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
faultNoExecute(); faultNoExecute();
} else if (eql(case, "fault-null")) { } else if (eql(case, "fault-null")) {
faultNull(); faultNull();
} else if (eql(case, "user")) {
userTest();
} else if (eql(case, "user-pf")) {
userPfTest();
} else if (eql(case, "init")) {
initTest(boot_info);
} else if (eql(case, "poweroff")) { } else if (eql(case, "poweroff")) {
powerTest(.off); powerTest(.off);
} else if (eql(case, "reboot")) { } else if (eql(case, "reboot")) {
@@ -702,6 +709,67 @@ fn smpRetryTest() void {
result(); result();
} }
// --- ring 3 (user mode) -----------------------------------------------------
/// The full ring-3 round trip: enter user mode, take syscalls and timer
/// interrupts from CPL 3, and come back. Preemption is disabled for the run —
/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate
/// (interrupts still fire and iretq back into ring 3, which is the point).
fn userTest() void {
log("DANOS-TEST-BEGIN: user\n", .{});
sched.setPreemption(false);
const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: {
log("DANOS-USER: run failed: {s}\n", .{@errorName(err)});
break :blk false;
};
sched.setPreemption(true);
check("user program ran and exited (ring-3 round trip)", ran);
check("two ping syscalls received", usermode.ping_count == 2);
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23);
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
result();
}
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
/// RIP. The fault report is the pass signal (matched by the harness); if the
/// read is somehow allowed the blob spins and the harness times out.
fn userPfTest() void {
log("DANOS-TEST-BEGIN: user-pf\n", .{});
sched.setPreemption(false);
_ = usermode.run(usermode.pfBlob()) catch {};
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
}
/// The full user-binary path: the bootloader read sbin/init off the boot
/// volume and handed it over; load it as a user ELF and run it in ring 3. The
/// same call the normal boot path makes — here with teeth.
fn initTest(boot_info: *const BootInfo) void {
log("DANOS-TEST-BEGIN: init\n", .{});
check("bootloader handed over sbin/init", boot_info.init_len != 0);
if (boot_info.init_len == 0) {
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
sched.setPreemption(false); // see userTest: pins the run to this core's rsp0
const code = usermode.runInitElf(image);
sched.setPreemption(true);
if (code) |c| {
check("init loaded, ran, and exited (user ELF path)", true);
check("init exited cleanly (code 0)", c == 0);
} else |err| {
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
check("init loaded, ran, and exited (user ELF path)", false);
}
check("init's write arrived intact", eql(usermode.write_buf[0..usermode.write_len], "init: hello from user space\n"));
check("write came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23);
result();
}
fn faultInvalidOpcode() void { fn faultInvalidOpcode() void {
log("DANOS-TEST-BEGIN: fault-ud\n", .{}); log("DANOS-TEST-BEGIN: fault-ud\n", .{});
asm volatile ("ud2"); asm volatile ("ud2");
+298
View File
@@ -0,0 +1,298 @@
//! Ring-3 execution — the isolation track's first rung (M1). Two entry points:
//! `run` executes a raw code blob (the user/user-pf test programs), and
//! `runInitElf` loads and runs a real user ELF (`/sbin/init`, handed over by
//! the bootloader). Both run at CPL 3 in the current (global) page tables:
//! frames are mapped user-accessible with W^X (code RO+X, data RW+NX), the CPU
//! drops privilege via iretq, and the program talks to the kernel only through
//! the `int $0x80` syscall gate. No processes or per-task address spaces yet —
//! this proves the privilege mechanisms (user descriptors, TSS.rsp0, the U/S
//! page bit, ring transitions) that M3 builds on.
//!
//! Caller contract (M1 limitations):
//! - enter/exit publishes TSS.rsp0 on the *current* core only, so the caller
//! must prevent migration for the duration — disable preemption around the
//! call (rsp0-per-context-switch arrives with real user tasks in M3).
//! - The kernel-side saved context (`user_saved_rsp` in isr.s) is a single
//! global: at most one core may be inside user mode at a time.
const std = @import("std");
const elf = std.elf;
const danos = @import("danos");
const arch = @import("arch");
const pmm = @import("pmm.zig");
const sched = @import("scheduler.zig");
const log = @import("log.zig");
const page_size = danos.page_size;
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
/// the identity map (low indices) and the vmm test address (index 128), so
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
/// An ELF image may occupy [code_virt, stack_virt); the stack page sits above.
pub const code_virt: u64 = 0x0000_7000_0000_0000;
pub const stack_virt: u64 = 0x0000_7000_0020_0000;
// The hand-assembled user program blobs (isr.s, .rodata).
const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" });
const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" });
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit.
pub fn helloBlob() []const u8 {
return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)];
}
/// The isolation-proof program: reads a kernel-only page, must #PF.
pub fn pfBlob() []const u8 {
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
}
/// What a ping syscall recorded — evidence for the test to assert on.
pub const Ping = struct { value: u64 = 0, cs: u64 = 0, ticks: u64 = 0 };
pub var pings = [_]Ping{.{}} ** 2;
pub var ping_count: usize = 0;
/// What write syscalls produced (accumulated), and the exit syscall's code.
pub var write_buf: [256]u8 = undefined;
pub var write_len: usize = 0;
pub var write_cs: u64 = 0;
pub var exit_code: u64 = 0;
/// The M1 syscall surface, dispatched on the saved user rax:
/// 0 = exit(code) — unwind back into the kernel context that entered
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
/// replaces this in M3; rax is written back as the return value already, since
/// isr_common restores user registers from the trap frame.
fn syscall(state: *arch.CpuState) void {
switch (state.rax) {
0 => {
exit_code = state.rdi;
arch.userExit();
},
1 => {
if (ping_count < pings.len) {
pings[ping_count] = .{ .value = state.rdi, .cs = state.cs, .ticks = arch.ticks() };
ping_count += 1;
}
state.rax = 0;
},
2 => {
// The pointer must lie inside the user region (image + stack) —
// kernel addresses and non-canonical values fall outside it, so the
// kernel-side read below can't be steered at kernel data. Length is
// checked first so the upper-bound subtraction can't underflow.
// Known gap (fine for a trusted init): a pointer into an *unmapped*
// hole in the region passes the check and the read #PFs -> on_fault
// halts — a self-DoS, not an isolation break. Copy-in with fault
// recovery is M3+ (with SMAP, once there's a reason to enable it).
const ptr = state.rdi;
const len = state.rsi;
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
const src: [*]const u8 = @ptrFromInt(ptr);
const n = @min(len, write_buf.len - write_len);
@memcpy(write_buf[write_len..][0..n], src[0..n]);
write_len += n;
write_cs = state.cs;
log.write("DANOS-INIT: ");
log.write(src[0..len]);
state.rax = len;
} else {
state.rax = @bitCast(@as(i64, -1));
}
},
else => state.rax = @bitCast(@as(i64, -1)),
}
}
/// Reset the recorded syscall evidence before a user-mode run.
fn resetRecords() void {
ping_count = 0;
write_len = 0;
write_cs = 0;
exit_code = 0;
}
pub const RunError = error{ ProgramTooBig, OutOfMemory };
/// Map `blob` at code_virt with a fresh user stack, drop to ring 3, and return
/// once the program exits via syscall 0. See the migration caveat in the module
/// doc. A program that faults instead never returns (on_fault halts the core).
pub fn run(blob: []const u8) RunError!void {
if (blob.len > page_size) return error.ProgramTooBig;
const code_frame = pmm.alloc() orelse return error.OutOfMemory;
const stack_frame = pmm.alloc() orelse {
pmm.free(code_frame);
return error.OutOfMemory;
};
// Fill the code frame through its identity mapping (supervisor RW): the
// user-facing mapping is read-only, and this also sidesteps CR0.WP/SMAP.
// The tail is padded with int3 so a stray jump traps instead of sliding.
const code: [*]u8 = @ptrFromInt(code_frame);
@memcpy(code[0..blob.len], blob);
@memset(code[blob.len..page_size], 0xCC);
arch.setSyscallHandler(syscall);
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
resetRecords();
arch.enterUser(sched.currentCpuIndex(), code_virt, stack_virt + page_size);
// Back via the exit syscall; the interrupt gate left IF clear.
arch.enableInterrupts();
arch.unmapPage(code_virt);
arch.unmapPage(stack_virt);
pmm.free(code_frame);
pmm.free(stack_frame);
}
// --- user ELF loading (/sbin/init) ------------------------------------------
pub const InitError = error{
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
BadEntry, // e_entry not inside an executable segment
ProgramTooBig, // more pages than the loader's budget
OutOfMemory,
};
const max_segments = 16;
const max_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
const Segment = struct {
vaddr: u64,
memsz: u64,
filesz: u64,
off: u64,
writable: bool,
executable: bool,
fn pages(self: Segment) u64 {
return (self.memsz + page_size - 1) / page_size;
}
};
/// Pages mapped for the running init image, for rollback if loading fails
/// midway. On success the mappings stay for the system's life — there's no
/// address-space teardown until real processes exist (M3).
var loaded = [_]struct { virt: u64, frame: u64 }{.{ .virt = 0, .frame = 0 }} ** (max_pages + 1);
var loaded_count: usize = 0;
/// Parse and validate every PT_LOAD before touching memory. Bounds are checked
/// against the image and the user region; segments must be page-aligned,
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
/// segment, mapped RO+NX).
fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!struct { count: usize, entry: u64 } {
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf;
if (ehdr.e_machine != .X86_64) return error.BadElf;
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
if (ehdr.e_phnum > max_segments) return error.BadElf;
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
var count: usize = 0;
var total_pages: u64 = 0;
for (0..ehdr.e_phnum) |i| {
const off = ehdr.e_phoff + i * ehdr.e_phentsize;
const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]);
if (phdr.p_type != elf.PT_LOAD) continue;
if (phdr.p_memsz == 0) continue;
if (phdr.p_vaddr % page_size != 0) return error.BadSegment;
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
// Inside the user image region, strictly below the stack page.
if (phdr.p_vaddr < code_virt) return error.BadSegment;
if (phdr.p_memsz > stack_virt - phdr.p_vaddr) return error.BadSegment;
const w = phdr.p_flags & elf.PF_W != 0;
const x = phdr.p_flags & elf.PF_X != 0;
if (w and x) return error.BadSegment; // W^X, even for init
const seg = Segment{
.vaddr = phdr.p_vaddr,
.memsz = phdr.p_memsz,
.filesz = phdr.p_filesz,
.off = phdr.p_offset,
.writable = w,
.executable = x,
};
// No overlap with any earlier segment (page-granular, since mapping is).
for (segs[0..count]) |other| {
const a_end = seg.vaddr + seg.pages() * page_size;
const b_end = other.vaddr + other.pages() * page_size;
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
}
total_pages += seg.pages();
if (total_pages > max_pages) return error.ProgramTooBig;
segs[count] = seg;
count += 1;
}
if (count == 0) return error.BadElf;
// The entry point must land inside an executable segment.
for (segs[0..count]) |seg| {
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
return .{ .count = count, .entry = ehdr.e_entry };
}
return error.BadEntry;
}
/// Map one page of a segment: fresh frame, zero + copy through the identity
/// mapping (the user mapping may be read-only), then the user-visible mapping
/// with the segment's W^X permissions. Records the page for rollback.
fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void {
const frame = pmm.alloc() orelse return error.OutOfMemory;
const dst: [*]u8 = @ptrFromInt(frame);
@memset(dst[0..page_size], 0);
const page_off = page_index * page_size;
if (page_off < seg.filesz) {
const n = @min(page_size, seg.filesz - page_off);
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
}
const virt = seg.vaddr + page_off;
arch.mapUserPage(virt, frame, seg.writable, seg.executable);
loaded[loaded_count] = .{ .virt = virt, .frame = frame };
loaded_count += 1;
}
fn unloadAll() void {
for (loaded[0..loaded_count]) |p| {
arch.unmapPage(p.virt);
pmm.free(p.frame);
}
loaded_count = 0;
}
/// Load a user ELF image, run it in ring 3 from its entry point, and return its
/// exit code. Same caller contract as `run` (preemption off, one core). On
/// success the user mappings are left in place — teardown comes with real
/// processes (M3); on a loading error everything is rolled back.
pub fn runInitElf(image: []const u8) InitError!u64 {
var segs: [max_segments]Segment = undefined;
const parsed = try parseSegments(image, &segs);
loaded_count = 0;
errdefer unloadAll();
for (segs[0..parsed.count]) |seg| {
for (0..seg.pages()) |i| try loadPage(image, seg, i);
}
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame };
loaded_count += 1;
arch.setSyscallHandler(syscall);
resetRecords();
arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size);
arch.enableInterrupts(); // the exit arrived through an interrupt gate
return exit_code;
}
+6
View File
@@ -109,4 +109,10 @@ pub const BootInfo = extern struct {
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob` /// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
/// field instead, so the kernel discovers devices without knowing what booted it. /// field instead, so the kernel discovers devices without knowing what booted it.
acpi_rsdp: u64 = 0, acpi_rsdp: u64 = 0,
/// The raw `/sbin/init` ELF image, read off the boot volume by the loader
/// into memory that survives the handoff (classified reserved, so the kernel
/// identity-maps it and never allocates over it). 0/0 = no init found — the
/// kernel boots without user space. Grows into a full initrd handoff later.
init_base: u64 = 0,
init_len: u64 = 0,
}; };
+20
View File
@@ -57,6 +57,8 @@ ARCHES = {
], ],
"efi_app": ("EFI/BOOT/BOOTX64.efi", "BOOTX64.efi"), # (dest in ESP, name in zig-out/bin) "efi_app": ("EFI/BOOT/BOOTX64.efi", "BOOTX64.efi"), # (dest in ESP, name in zig-out/bin)
"kernel": ("kernel", "kernel"), "kernel": ("kernel", "kernel"),
# Further files shipped on the ESP: the init user program.
"extra": [("sbin/init", "init")],
# Built as a function so we can splice in per-run paths. # Built as a function so we can splice in per-run paths.
"qemu_args": lambda a, esp, vars_fd, serial: [ "qemu_args": lambda a, esp, vars_fd, serial: [
"-machine", "q35", "-m", "128M", "-machine", "q35", "-m", "128M",
@@ -148,6 +150,21 @@ CASES = [
"expect": r"page fault \(vector 14\)", "expect": r"page fault \(vector 14\)",
"fail": r"NX not enforced"}, "fail": r"NX not enforced"},
{"name": "fault-null", "expect": r"page fault \(vector 14\)"}, {"name": "fault-null", "expect": r"page fault \(vector 14\)"},
# Ring 3: a user program runs at CPL 3, makes int 0x80 syscalls, survives
# timer interrupts, and exits back into the kernel.
{"name": "user",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
# 0x5 (present|user) at the user RIP. ([\s\S] spans lines; `.` doesn't.)
{"name": "user-pf",
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*RIP\s*: 0x00007000000000",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The real user binary: the bootloader ships sbin/init off the ESP, the
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
{"name": "init",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the # The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
# pre-transition marker; the FAIL line only appears if the transition didn't take. # pre-transition marker; the FAIL line only appears if the transition didn't take.
{"name": "poweroff", {"name": "poweroff",
@@ -179,6 +196,9 @@ def make_esp(arch):
os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True) os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True)
shutil.copy(os.path.join(REPO, "zig-out", "bin", efi_name), os.path.join(esp, efi_dest)) shutil.copy(os.path.join(REPO, "zig-out", "bin", efi_name), os.path.join(esp, efi_dest))
shutil.copy(os.path.join(REPO, "zig-out", "bin", kern_name), os.path.join(esp, kern_dest)) shutil.copy(os.path.join(REPO, "zig-out", "bin", kern_name), os.path.join(esp, kern_dest))
for dest, name in arch.get("extra", []):
os.makedirs(os.path.join(esp, os.path.dirname(dest)), exist_ok=True)
shutil.copy(os.path.join(REPO, "zig-out", "bin", name), os.path.join(esp, dest))
return esp return esp