isolation M1: ring 3 + a real /sbin/init, end to end
Ring 3 works: user GDT descriptors (sysret-ready layout), TSS.rsp0, U/S-bit user mappings (W^X preserved), an int 0x80 syscall gate with a mutable trap frame, and a setjmp-style enter/exit path. /sbin/init is a real freestanding Zig binary built from sbin/, shipped on the ESP, loaded by the bootloader (BootInfo.init_base/len), validated and mapped by an in-kernel user-ELF loader, and run at CPL 3 — syscalls: exit, ping, write. Tests: user, user-pf (U/S isolation proof, error code 0x5), init. Suite 27/27. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
7501bd1703
commit
546dd44a2a
@@ -145,6 +145,31 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
b.installArtifact(exe);
|
||||
|
||||
// --- /sbin/init: the first user-space program ---
|
||||
// Its own tiny freestanding binary, linked at a fixed address inside the
|
||||
// kernel's user region (usermode.zig) and started in ring 3 by the kernel's
|
||||
// user-ELF loader. `.large` because the image base is above 4 GiB — small/
|
||||
// medium code models emit 32-bit absolute relocations that can't reach.
|
||||
// Pinned to ReleaseSmall: the user region gives it a 2 MiB budget and its
|
||||
// size has no reason to track the kernel's optimize mode.
|
||||
const init_exe = b.addExecutable(.{
|
||||
.name = "init",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("sbin/init.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = true,
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
}),
|
||||
});
|
||||
init_exe.setLinkerScript(b.path("sbin/linker.ld"));
|
||||
init_exe.entry = .{ .symbol_name = "_start" };
|
||||
init_exe.image_base = 0x7000_0000_0000;
|
||||
b.installArtifact(init_exe);
|
||||
|
||||
// Boot methods live in src/boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
@@ -203,6 +228,10 @@ pub fn build(b: *std.Build) void {
|
||||
const kernel_install = b.addInstallArtifact(exe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
||||
});
|
||||
// The bootloader loads init from sbin/init on the same volume.
|
||||
const init_install = b.addInstallArtifact(init_exe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp/sbin" } },
|
||||
});
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
@@ -239,6 +268,7 @@ pub fn build(b: *std.Build) void {
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
run_efi.step.dependOn(&efi_install.step);
|
||||
run_efi.step.dependOn(&kernel_install.step);
|
||||
run_efi.step.dependOn(&init_install.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
+7
-1
@@ -1,5 +1,11 @@
|
||||
# System Calls
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
|
||||
> **Status:** danos currently has a placeholder M1 surface behind an `int 0x80`
|
||||
> gate — `0 = exit(code)`, `1 = ping(value)`, `2 = write(ptr, len)` (see
|
||||
> `src/kernel/usermode.zig`, used by `sbin/init.zig`). It exists to prove the
|
||||
> ring transition; the microkernel set below replaces it (via `syscall`/`sysret`)
|
||||
> when processes land (M3).
|
||||
|
||||
## The Mechanism of a Syscall
|
||||
|
||||
|
||||
+8
-3
@@ -81,11 +81,16 @@ prerequisites.
|
||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md).
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), and
|
||||
**ring 3**: user GDT/TSS plumbing, U/S-bit mappings, an `int 0x80` syscall gate, and
|
||||
`/sbin/init` — a real user ELF built from `sbin/`, shipped on the boot volume, loaded
|
||||
by the kernel, run at CPL 3 — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
|
||||
ring 3, per-process page tables). The substrate everything else needs. *Next, and a
|
||||
prerequisite for the resilience and driver tracks.*
|
||||
ring 3, per-process page tables). The substrate everything else needs. *In
|
||||
progress: ring 3 + a loaded `/sbin/init` work (M1); next the higher-half move (M2),
|
||||
then per-process address spaces + `syscall`/`sysret` + init as a real schedulable
|
||||
process (M3), then ELF/initrd generalisation (M4).*
|
||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||
See [resilience.md](resilience.md).
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
//! /sbin/init — the first user-space program. Built as its own freestanding
|
||||
//! binary (see build.zig), shipped on the boot volume at sbin/init, loaded by
|
||||
//! the bootloader, and started in ring 3 by the kernel's user-ELF loader
|
||||
//! (src/kernel/usermode.zig). It talks to the kernel only through the
|
||||
//! `int $0x80` syscall gate.
|
||||
//!
|
||||
//! Today it just proves the path — say hello, exit — and grows into the real
|
||||
//! init (service supervision) once processes are schedulable (M3).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// The M1 syscall numbers (usermode.zig): 0 = exit(code), 2 = write(ptr, len).
|
||||
const sys_exit = 0;
|
||||
const sys_write = 2;
|
||||
|
||||
fn syscall2(n: u64, a: u64, b: u64) u64 {
|
||||
return asm volatile ("int $0x80"
|
||||
: [ret] "={rax}" (-> u64),
|
||||
: [n] "{rax}" (n),
|
||||
[a] "{rdi}" (a),
|
||||
[b] "{rsi}" (b),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
fn write(msg: []const u8) void {
|
||||
_ = syscall2(sys_write, @intFromPtr(msg.ptr), msg.len);
|
||||
}
|
||||
|
||||
fn exit(code: u64) noreturn {
|
||||
_ = syscall2(sys_exit, code, 0);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Entry. Naked: the kernel enters with rsp 16-aligned, but a SysV function
|
||||
/// expects rsp ≡ 8 (mod 16) on entry (as if reached by `call`) — so re-enter
|
||||
/// the ABI with an actual call. The trap after is unreachable.
|
||||
pub export fn _start() callconv(.naked) noreturn {
|
||||
asm volatile (
|
||||
\\call init_main
|
||||
\\ud2
|
||||
);
|
||||
}
|
||||
|
||||
export fn init_main() callconv(.c) noreturn {
|
||||
write("init: hello from user space\n");
|
||||
exit(0);
|
||||
}
|
||||
|
||||
/// No runtime to unwind into — report the panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,45 @@
|
||||
/* /sbin/init link layout.
|
||||
*
|
||||
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||
* base, so the entry point comes from e_entry, not the base address.
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space; the loader zeroes the
|
||||
* filesz..memsz gap. */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,10 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
||||
|
||||
/// Path of the init program on the boot volume (UEFI paths use backslashes;
|
||||
/// the FAT driver walks the components itself, so no directory dance needed).
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("sbin\\init");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
const seek_end = 0xffff_ffff_ffff_ffff;
|
||||
@@ -54,6 +58,13 @@ fn boot() !noreturn {
|
||||
|
||||
const entry = try loadKernel(bs, &boot_info);
|
||||
|
||||
// Best effort: a volume without sbin/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_info) catch |err| {
|
||||
log("danos: no sbin/init (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
log("danos: kernel loaded, exiting boot services\r\n");
|
||||
boot_info.memory_map = try exitBootServices(bs);
|
||||
|
||||
@@ -198,6 +209,43 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
||||
return loadElf(bs, image, boot_info);
|
||||
}
|
||||
|
||||
/// Read the init program (sbin/init) into memory that outlives the loader and
|
||||
/// record it in the handoff. The pool buffer is deliberately NOT freed: it's
|
||||
/// LoaderData, which the memory-map conversion classifies as reserved, so the
|
||||
/// kernel identity-maps it and reads the ELF from there. The kernel does the
|
||||
/// loading itself (into ring-3 mappings) — the loader just ferries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !void {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
|
||||
return error.NoFileSystem;
|
||||
|
||||
const root = try fs.openVolume();
|
||||
defer _ = root.close() catch {};
|
||||
|
||||
const file = try root.open(init_file_name, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
|
||||
try file.setPosition(seek_end);
|
||||
const size: usize = @intCast(try file.getPosition());
|
||||
try file.setPosition(0);
|
||||
if (size == 0) return error.EmptyFile;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||
|
||||
var read_total: usize = 0;
|
||||
while (read_total < size) {
|
||||
const n = try file.read(image[read_total..]);
|
||||
if (n == 0) return error.UnexpectedEof;
|
||||
read_total += n;
|
||||
}
|
||||
|
||||
boot_info.init_base = @intFromPtr(image.ptr);
|
||||
boot_info.init_len = size;
|
||||
log("danos: sbin/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
/// record each segment's layout so the kernel can re-map itself with the right
|
||||
/// permissions.
|
||||
|
||||
@@ -76,6 +76,45 @@ pub fn unmapPage(virt: u64) void {
|
||||
paging.unmap(virt);
|
||||
}
|
||||
|
||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||
/// W^X: code read-only + executable, data writable + no-execute.
|
||||
pub fn mapUserPage(virt: u64, phys: u64, writable: bool, executable: bool) void {
|
||||
paging.mapUser(virt, phys, writable, executable);
|
||||
}
|
||||
|
||||
// --- ring 3 entry/exit -----------------------------------------------------
|
||||
|
||||
/// Drop to ring 3 at `rip` on `rsp` (defined in isr.s). Saves the kernel context,
|
||||
/// publishes the kernel stack pointer through `rsp0_slot` (this core's TSS.rsp0,
|
||||
/// so ring-3 interrupts land on a good stack), builds an iretq frame with the
|
||||
/// user selectors, and iretq's. "Returns" only when the user program triggers
|
||||
/// the exit path (user_exit_to_kernel).
|
||||
extern fn enter_user(rip: u64, rsp: u64, rsp0_slot: *align(4) u64) callconv(.c) void;
|
||||
|
||||
/// Abandon the in-flight ring-3 trap context and resume the kernel as if
|
||||
/// `enter_user` had returned (defined in isr.s). Called by the exit syscall.
|
||||
extern fn user_exit_to_kernel() callconv(.c) noreturn;
|
||||
|
||||
/// Run user code at `rip` with stack `rsp` on this core (`cpu` = the caller's CPU
|
||||
/// index; the arch layer can't ask the scheduler). Returns after the user program
|
||||
/// exits via syscall. Interrupts are disabled on return (the exit arrives through
|
||||
/// an interrupt gate) — the caller re-enables.
|
||||
pub fn enterUser(cpu: usize, rip: u64, rsp: u64) void {
|
||||
enter_user(rip, rsp, tss.rsp0Ptr(cpu));
|
||||
}
|
||||
|
||||
/// Never returns to the user program: unwind to the kernel context that called
|
||||
/// `enterUser`. For the exit syscall's handler.
|
||||
pub fn userExit() noreturn {
|
||||
user_exit_to_kernel();
|
||||
}
|
||||
|
||||
/// Register the handler for the ring-3 syscall gate (int 0x80, vector 128). The
|
||||
/// handler may write the trap frame (e.g. rax as the return value).
|
||||
pub fn setSyscallHandler(handler: *const fn (*idt.CpuState) void) void {
|
||||
idt.setSyscallHandler(handler);
|
||||
}
|
||||
|
||||
/// CR3 holds the physical address of the active top-level page table.
|
||||
pub fn readCr3() u64 {
|
||||
return asm volatile ("mov %%cr3, %[out]"
|
||||
@@ -84,9 +123,11 @@ pub fn readCr3() u64 {
|
||||
}
|
||||
|
||||
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
|
||||
/// data pointer (there's no user mode yet, so no `swapgs` dance — GS base is always
|
||||
/// the running core's per-CPU block). Set once per core during bring-up, after the
|
||||
/// GDT is loaded (loading a GS *selector* would otherwise clobber this base).
|
||||
/// data pointer. Because it's always read back via `rdmsr` (never gs-relative
|
||||
/// addressing), no `swapgs` dance is needed even with user mode: the MSR is
|
||||
/// privileged, ring 3 can't touch it, and its value is unaffected by ring
|
||||
/// transitions. Set once per core during bring-up, after the GDT is loaded
|
||||
/// (loading a GS *selector* would otherwise clobber this base).
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
|
||||
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
||||
//! reference a code selector — so we install our own flat GDT with known
|
||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
||||
//! the firmware left in place.
|
||||
//! selectors (0x08/0x10 kernel code/data, 0x18/0x20 user data/code for ring 3)
|
||||
//! rather than trusting whatever the firmware left in place.
|
||||
//!
|
||||
//! The code/data descriptors are identical on every core, but the **TSS descriptor
|
||||
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
|
||||
@@ -14,20 +14,35 @@ const config = @import("config");
|
||||
/// Selectors into the table (index * 8). Same on every core's GDT.
|
||||
pub const kernel_code = 0x08;
|
||||
pub const kernel_data = 0x10;
|
||||
pub const tss_selector = 0x18;
|
||||
pub const user_data = 0x18;
|
||||
pub const user_code = 0x20;
|
||||
pub const tss_selector = 0x28;
|
||||
|
||||
/// Ring-3 selectors as loaded from user mode: RPL 3 or'd in. The user *data*
|
||||
/// descriptor is load-bearing even in long mode — iretq to CPL 3 with a null SS
|
||||
/// raises #GP(0).
|
||||
pub const user_code_rpl3 = user_code | 3;
|
||||
pub const user_data_rpl3 = user_data | 3;
|
||||
|
||||
const max_cpus = config.max_cpus;
|
||||
const entries = 5; // null, code, data, TSS-low, TSS-high
|
||||
const entries = 7; // null, kcode, kdata, udata, ucode, TSS-low, TSS-high
|
||||
|
||||
/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor,
|
||||
/// The shared descriptors (slots 0-4); slots 5-6 hold this core's TSS descriptor,
|
||||
/// filled in per core by `setTssFor`.
|
||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// kernel code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// kernel data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// user data: present, ring 3, writable -> 0x00CFF200_0000FFFF
|
||||
/// user code: present, ring 3, executable, readable, L=1 -> 0x00AFFA00_0000FFFF
|
||||
/// User data sits below user code so a future SYSRET works unchanged: it loads
|
||||
/// CS = STAR.SYSRET_CS + 16 and SS = STAR.SYSRET_CS + 8, so with SYSRET_CS = 0x10
|
||||
/// those land on 0x20 (user code) and 0x18 (user data).
|
||||
const template = [entries]u64{
|
||||
0, // null descriptor (required)
|
||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
||||
0x00CF92000000FFFF, // kernel data (0x10)
|
||||
0, // TSS descriptor low (0x18)
|
||||
0x00CFF2000000FFFF, // user data (0x18)
|
||||
0x00AFFA000000FFFF, // user code (0x20)
|
||||
0, // TSS descriptor low (0x28)
|
||||
0, // TSS descriptor high
|
||||
};
|
||||
|
||||
@@ -38,13 +53,13 @@ var gdts = [_][entries]u64{template} ** max_cpus;
|
||||
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
|
||||
/// TSS. Write it into that core's GDT before it loads the TSS selector.
|
||||
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
|
||||
gdts[cpu][3] = (limit & 0xFFFF) |
|
||||
gdts[cpu][5] = (limit & 0xFFFF) |
|
||||
((base & 0xFFFF) << 16) |
|
||||
(((base >> 16) & 0xFF) << 32) |
|
||||
(@as(u64, 0x89) << 40) |
|
||||
(((limit >> 16) & 0xF) << 48) |
|
||||
(((base >> 24) & 0xFF) << 56);
|
||||
gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF;
|
||||
gdts[cpu][6] = (base >> 32) & 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
||||
|
||||
@@ -26,6 +26,19 @@ pub fn setHandler(vector: usize, handler: Handler) void {
|
||||
handlers[vector] = handler;
|
||||
}
|
||||
|
||||
/// The ring-3 syscall gate's vector (`int $0x80`, the classic choice — well away
|
||||
/// from the device range) and its handler. Unlike device handlers, a syscall
|
||||
/// handler gets the (mutable) trap frame: it reads its arguments from the saved
|
||||
/// user registers and writes rax as the return value, which isr_common then
|
||||
/// restores into the user context.
|
||||
pub const syscall_vector = 128;
|
||||
|
||||
var syscall_handler: ?*const fn (*CpuState) void = null;
|
||||
|
||||
pub fn setSyscallHandler(handler: *const fn (*CpuState) void) void {
|
||||
syscall_handler = handler;
|
||||
}
|
||||
|
||||
/// The register + trap frame the ISR stubs build on the stack, laid out so the
|
||||
/// lowest address (where RSP points when we call the handler) is the first field.
|
||||
/// See the push order in `isrCommon` below.
|
||||
@@ -128,6 +141,13 @@ pub fn init() void {
|
||||
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
|
||||
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
|
||||
idt[8].ist = tss.double_fault_ist;
|
||||
// The syscall gate. Installed outside the 0..gate_count loop (stubs 48-127
|
||||
// don't exist) and with DPL 3 — without it, `int $0x80` from ring 3 is a
|
||||
// #GP. An interrupt gate (not trap): IF is cleared for the handler, which
|
||||
// the ring-3 exit path relies on.
|
||||
const syscall_stub = @extern(*const anyopaque, .{ .name = "isr128" });
|
||||
setGate(syscall_vector, @intFromPtr(syscall_stub));
|
||||
idt[syscall_vector].flags = 0xEE; // present, DPL 3, 64-bit interrupt gate
|
||||
loadOnThisCpu();
|
||||
}
|
||||
|
||||
@@ -145,9 +165,12 @@ pub fn loadOnThisCpu() void {
|
||||
/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the
|
||||
/// assembly stubs can `call` it by name. Exceptions are terminal; device
|
||||
/// interrupts run their handler, get acknowledged, and return.
|
||||
export fn interruptDispatch(state: *const CpuState) callconv(.c) void {
|
||||
export fn interruptDispatch(state: *CpuState) callconv(.c) void {
|
||||
if (state.vector < 32) {
|
||||
on_fault(state); // CPU exception — never returns
|
||||
} else if (state.vector == syscall_vector) {
|
||||
// Software interrupt from ring 3 — no LAPIC ISR bit is set, so no EOI.
|
||||
if (syscall_handler) |handler| handler(state);
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
// Acknowledge before running the handler: a handler that switches tasks
|
||||
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
|
||||
|
||||
@@ -76,6 +76,96 @@ task_trampoline:
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# --- ring 3 entry/exit ------------------------------------------------------
|
||||
|
||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
|
||||
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
|
||||
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
|
||||
# grow safely into it — then builds the 5-word iretq frame with the user
|
||||
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
|
||||
# timer keeps running in user mode.
|
||||
.global enter_user
|
||||
enter_user:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
|
||||
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
|
||||
push $0x1B # user SS (0x18 | RPL 3)
|
||||
push %rsi # user RSP
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP
|
||||
iretq
|
||||
|
||||
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
|
||||
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
|
||||
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
|
||||
# and stay off — the Zig caller re-enables.
|
||||
.global user_exit_to_kernel
|
||||
user_exit_to_kernel:
|
||||
mov user_saved_rsp(%rip), %rsp
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret
|
||||
|
||||
.section .bss
|
||||
.balign 8
|
||||
user_saved_rsp:
|
||||
.skip 8
|
||||
.text
|
||||
|
||||
# --- user-mode test programs -------------------------------------------------
|
||||
# Hand-assembled ring-3 blobs, copied by the kernel onto a user-mapped page and
|
||||
# entered via enter_user. Position-independent (immediates and short jumps only).
|
||||
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
|
||||
# from the user mapping.
|
||||
.section .rodata
|
||||
|
||||
# The hello program: ping syscall (rax=1) with a value, a delay loop long enough
|
||||
# for several timer ticks to land while at CPL 3 (proving interrupt-from-user +
|
||||
# iretq-back), a second ping, then exit (rax=0). The trailing jmp is a safety
|
||||
# net in case exit ever returns.
|
||||
.global user_prog_start
|
||||
.global user_prog_end
|
||||
user_prog_start:
|
||||
mov $1, %rax
|
||||
mov $0xC0DE, %rdi
|
||||
int $0x80
|
||||
mov $50000000, %rcx # ~50M iterations: tens of ms even under TCG
|
||||
1: dec %rcx
|
||||
jnz 1b
|
||||
mov $1, %rax
|
||||
mov $0xBEEF, %rdi
|
||||
int $0x80
|
||||
mov $0, %rax
|
||||
int $0x80
|
||||
2: jmp 2b
|
||||
user_prog_end:
|
||||
|
||||
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
||||
# page is unconditionally mapped supervisor (paging.zig), so this must take a
|
||||
# #PF with error code 0x5 (present | user) before any access happens. The
|
||||
# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax`
|
||||
# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page.
|
||||
.global user_pf_start
|
||||
.global user_pf_end
|
||||
user_pf_start:
|
||||
mov $0xFEE00000, %ecx
|
||||
mov (%rcx), %rax
|
||||
1: jmp 1b
|
||||
user_pf_end:
|
||||
|
||||
.text
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
@@ -145,6 +235,10 @@ STUB_NOERR 45
|
||||
STUB_NOERR 46
|
||||
STUB_NOERR 47
|
||||
|
||||
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
|
||||
# vector; dispatched specially in interruptDispatch.
|
||||
STUB_NOERR 128
|
||||
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
|
||||
@@ -18,6 +18,7 @@ const page_size = danos.page_size;
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
const writable: u64 = 1 << 1;
|
||||
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
@@ -128,6 +129,34 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
||||
/// pre-existing): ring-3 access requires U at *every* level, and `descend` leaves
|
||||
/// existing entries untouched. Only used under user-exclusive virtual ranges, so
|
||||
/// no kernel mapping's protection is widened (the leaf still governs).
|
||||
fn descendUser(entry: *u64) u64 {
|
||||
const table = descend(entry);
|
||||
entry.* |= user;
|
||||
return table;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virt` -> `phys` accessible from ring 3. W^X is the
|
||||
/// caller's contract: code pages are read-only + executable, data pages are
|
||||
/// writable + no-execute. `virt` must lie in a user-exclusive region (see
|
||||
/// `descendUser`).
|
||||
pub fn mapUser(virt: u64, phys: u64, writable_page: bool, executable: bool) void {
|
||||
var flags: u64 = present | user;
|
||||
if (writable_page) flags |= writable;
|
||||
if (!executable) flags |= no_execute;
|
||||
const pml4e = &tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
||||
const pdpt = descendUser(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
||||
const pd = descendUser(pdpte);
|
||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
||||
const pt = descendUser(pde);
|
||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags;
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
/// Whether `virt` is currently mapped **executable** — present with the NX bit
|
||||
/// clear. Walks the 4-level tables (all danos mappings are 4 KiB, so no huge-page
|
||||
/// case). Returns false if unmapped. Used for W^X checks in tests.
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
|
||||
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
|
||||
//! switches to that stack when the exception fires — no matter how broken the
|
||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
||||
//! that happens *because* the current stack is unusable still lands on solid
|
||||
//! ground instead of triple-faulting.
|
||||
//! Task State Segment and its interrupt stacks. In long mode the TSS has two
|
||||
//! jobs. First, the Interrupt Stack Table: an IDT gate can name an IST entry,
|
||||
//! and the CPU switches to that stack when the exception fires — no matter how
|
||||
//! broken the interrupted stack was. We use IST1 for the double-fault handler,
|
||||
//! so a fault that happens *because* the current stack is unusable still lands
|
||||
//! on solid ground instead of triple-faulting. Second, rsp0: the kernel stack
|
||||
//! the CPU switches to when an interrupt arrives from ring 3 (published by the
|
||||
//! user-mode entry path via `rsp0Ptr`).
|
||||
//!
|
||||
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
|
||||
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
|
||||
@@ -50,6 +52,16 @@ var ap_ist_top = [_]usize{0} ** max_cpus; // per-AP IST stack top (0 = BSP / not
|
||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||
|
||||
/// Address of core `cpu`'s rsp0 slot — the kernel stack the CPU switches to on a
|
||||
/// ring-3 -> ring-0 interrupt. Computed as base + 4 (rsp0's architectural offset,
|
||||
/// which is why the pointer is only 4-aligned) rather than `&t.rsp0`, which on a
|
||||
/// packed struct would be an unaligned bit-pointer type. The ring-3 entry path
|
||||
/// (enter_user in isr.s) writes the current kernel stack pointer through this
|
||||
/// before dropping to user mode.
|
||||
pub fn rsp0Ptr(cpu: usize) *align(4) u64 {
|
||||
return @ptrFromInt(@intFromPtr(&tss_table[cpu]) + 4);
|
||||
}
|
||||
|
||||
/// Record the top of the IST stack the kernel allocated for AP `cpu`. Called on the
|
||||
/// BSP before waking that core; read by the core's own `setupThisCpu`.
|
||||
pub fn setApIstStack(cpu: usize, top: usize) void {
|
||||
|
||||
+20
-1
@@ -7,6 +7,7 @@ const log = @import("log.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const usermode = @import("usermode.zig");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
@@ -245,7 +246,25 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
log.checkpoint(cp_running);
|
||||
status("kernel initialised.\n");
|
||||
|
||||
// TODO: init process
|
||||
// Hand over to user space: run /sbin/init (read off the boot volume by the
|
||||
// loader) in ring 3. Preemption is off for the run — the M1 user-mode path
|
||||
// publishes *this* core's TSS.rsp0 and must not migrate (the flag is
|
||||
// global, so the system goes cooperative meanwhile; the other cores are
|
||||
// idle). init becomes a real schedulable process in M3.
|
||||
if (boot_info.init_len != 0) {
|
||||
status("starting /sbin/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
|
||||
scheduler.setPreemption(false);
|
||||
const code = usermode.runInitElf(image);
|
||||
scheduler.setPreemption(true);
|
||||
if (code) |c| {
|
||||
statusPrint("/sbin/init exited with code {d}.\n", .{c});
|
||||
} else |err| {
|
||||
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
} else {
|
||||
status("no /sbin/init on the boot volume.\n");
|
||||
}
|
||||
|
||||
status("\nnothing left to do; halting CPU.\n");
|
||||
|
||||
|
||||
@@ -17,6 +17,7 @@ const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const usermode = @import("usermode.zig");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
||||
@@ -92,6 +93,12 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
faultNoExecute();
|
||||
} else if (eql(case, "fault-null")) {
|
||||
faultNull();
|
||||
} else if (eql(case, "user")) {
|
||||
userTest();
|
||||
} else if (eql(case, "user-pf")) {
|
||||
userPfTest();
|
||||
} else if (eql(case, "init")) {
|
||||
initTest(boot_info);
|
||||
} else if (eql(case, "poweroff")) {
|
||||
powerTest(.off);
|
||||
} else if (eql(case, "reboot")) {
|
||||
@@ -702,6 +709,67 @@ fn smpRetryTest() void {
|
||||
result();
|
||||
}
|
||||
|
||||
// --- ring 3 (user mode) -----------------------------------------------------
|
||||
|
||||
/// The full ring-3 round trip: enter user mode, take syscalls and timer
|
||||
/// interrupts from CPL 3, and come back. Preemption is disabled for the run —
|
||||
/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate
|
||||
/// (interrupts still fire and iretq back into ring 3, which is the point).
|
||||
fn userTest() void {
|
||||
log("DANOS-TEST-BEGIN: user\n", .{});
|
||||
sched.setPreemption(false);
|
||||
const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: {
|
||||
log("DANOS-USER: run failed: {s}\n", .{@errorName(err)});
|
||||
break :blk false;
|
||||
};
|
||||
sched.setPreemption(true);
|
||||
|
||||
check("user program ran and exited (ring-3 round trip)", ran);
|
||||
check("two ping syscalls received", usermode.ping_count == 2);
|
||||
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
|
||||
check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23);
|
||||
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
|
||||
result();
|
||||
}
|
||||
|
||||
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
|
||||
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
|
||||
/// RIP. The fault report is the pass signal (matched by the harness); if the
|
||||
/// read is somehow allowed the blob spins and the harness times out.
|
||||
fn userPfTest() void {
|
||||
log("DANOS-TEST-BEGIN: user-pf\n", .{});
|
||||
sched.setPreemption(false);
|
||||
_ = usermode.run(usermode.pfBlob()) catch {};
|
||||
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
||||
}
|
||||
|
||||
/// The full user-binary path: the bootloader read sbin/init off the boot
|
||||
/// volume and handed it over; load it as a user ELF and run it in ring 3. The
|
||||
/// same call the normal boot path makes — here with teeth.
|
||||
fn initTest(boot_info: *const BootInfo) void {
|
||||
log("DANOS-TEST-BEGIN: init\n", .{});
|
||||
check("bootloader handed over sbin/init", boot_info.init_len != 0);
|
||||
if (boot_info.init_len == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_info.init_base))[0..boot_info.init_len];
|
||||
sched.setPreemption(false); // see userTest: pins the run to this core's rsp0
|
||||
const code = usermode.runInitElf(image);
|
||||
sched.setPreemption(true);
|
||||
|
||||
if (code) |c| {
|
||||
check("init loaded, ran, and exited (user ELF path)", true);
|
||||
check("init exited cleanly (code 0)", c == 0);
|
||||
} else |err| {
|
||||
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
||||
check("init loaded, ran, and exited (user ELF path)", false);
|
||||
}
|
||||
check("init's write arrived intact", eql(usermode.write_buf[0..usermode.write_len], "init: hello from user space\n"));
|
||||
check("write came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23);
|
||||
result();
|
||||
}
|
||||
|
||||
fn faultInvalidOpcode() void {
|
||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
||||
asm volatile ("ud2");
|
||||
|
||||
@@ -0,0 +1,298 @@
|
||||
//! Ring-3 execution — the isolation track's first rung (M1). Two entry points:
|
||||
//! `run` executes a raw code blob (the user/user-pf test programs), and
|
||||
//! `runInitElf` loads and runs a real user ELF (`/sbin/init`, handed over by
|
||||
//! the bootloader). Both run at CPL 3 in the current (global) page tables:
|
||||
//! frames are mapped user-accessible with W^X (code RO+X, data RW+NX), the CPU
|
||||
//! drops privilege via iretq, and the program talks to the kernel only through
|
||||
//! the `int $0x80` syscall gate. No processes or per-task address spaces yet —
|
||||
//! this proves the privilege mechanisms (user descriptors, TSS.rsp0, the U/S
|
||||
//! page bit, ring transitions) that M3 builds on.
|
||||
//!
|
||||
//! Caller contract (M1 limitations):
|
||||
//! - enter/exit publishes TSS.rsp0 on the *current* core only, so the caller
|
||||
//! must prevent migration for the duration — disable preemption around the
|
||||
//! call (rsp0-per-context-switch arrives with real user tasks in M3).
|
||||
//! - The kernel-side saved context (`user_saved_rsp` in isr.s) is a single
|
||||
//! global: at most one core may be inside user mode at a time.
|
||||
|
||||
const std = @import("std");
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const pmm = @import("pmm.zig");
|
||||
const sched = @import("scheduler.zig");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||
/// setting the U/S bit on its intermediate tables widens no kernel mapping.
|
||||
/// An ELF image may occupy [code_virt, stack_virt); the stack page sits above.
|
||||
pub const code_virt: u64 = 0x0000_7000_0000_0000;
|
||||
pub const stack_virt: u64 = 0x0000_7000_0020_0000;
|
||||
|
||||
// The hand-assembled user program blobs (isr.s, .rodata).
|
||||
const prog_start = @extern([*]const u8, .{ .name = "user_prog_start" });
|
||||
const prog_end = @extern([*]const u8, .{ .name = "user_prog_end" });
|
||||
const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" });
|
||||
const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" });
|
||||
|
||||
/// The ring-3 hello program: ping(0xC0DE), delay loop, ping(0xBEEF), exit.
|
||||
pub fn helloBlob() []const u8 {
|
||||
return prog_start[0 .. @intFromPtr(prog_end) - @intFromPtr(prog_start)];
|
||||
}
|
||||
|
||||
/// The isolation-proof program: reads a kernel-only page, must #PF.
|
||||
pub fn pfBlob() []const u8 {
|
||||
return pf_start[0 .. @intFromPtr(pf_end) - @intFromPtr(pf_start)];
|
||||
}
|
||||
|
||||
/// What a ping syscall recorded — evidence for the test to assert on.
|
||||
pub const Ping = struct { value: u64 = 0, cs: u64 = 0, ticks: u64 = 0 };
|
||||
pub var pings = [_]Ping{.{}} ** 2;
|
||||
pub var ping_count: usize = 0;
|
||||
|
||||
/// What write syscalls produced (accumulated), and the exit syscall's code.
|
||||
pub var write_buf: [256]u8 = undefined;
|
||||
pub var write_len: usize = 0;
|
||||
pub var write_cs: u64 = 0;
|
||||
pub var exit_code: u64 = 0;
|
||||
|
||||
/// The M1 syscall surface, dispatched on the saved user rax:
|
||||
/// 0 = exit(code) — unwind back into the kernel context that entered
|
||||
/// 1 = ping(value) — record rdi + the caller's CS + the current tick count
|
||||
/// 2 = write(ptr, len) — log bytes from user memory, prefixed DANOS-INIT:
|
||||
/// The real microkernel ABI (IPC_Call/IPC_ReplyWait/Yield, docs/syscall.md)
|
||||
/// replaces this in M3; rax is written back as the return value already, since
|
||||
/// isr_common restores user registers from the trap frame.
|
||||
fn syscall(state: *arch.CpuState) void {
|
||||
switch (state.rax) {
|
||||
0 => {
|
||||
exit_code = state.rdi;
|
||||
arch.userExit();
|
||||
},
|
||||
1 => {
|
||||
if (ping_count < pings.len) {
|
||||
pings[ping_count] = .{ .value = state.rdi, .cs = state.cs, .ticks = arch.ticks() };
|
||||
ping_count += 1;
|
||||
}
|
||||
state.rax = 0;
|
||||
},
|
||||
2 => {
|
||||
// The pointer must lie inside the user region (image + stack) —
|
||||
// kernel addresses and non-canonical values fall outside it, so the
|
||||
// kernel-side read below can't be steered at kernel data. Length is
|
||||
// checked first so the upper-bound subtraction can't underflow.
|
||||
// Known gap (fine for a trusted init): a pointer into an *unmapped*
|
||||
// hole in the region passes the check and the read #PFs -> on_fault
|
||||
// halts — a self-DoS, not an isolation break. Copy-in with fault
|
||||
// recovery is M3+ (with SMAP, once there's a reason to enable it).
|
||||
const ptr = state.rdi;
|
||||
const len = state.rsi;
|
||||
if (len <= write_buf.len and ptr >= code_virt and ptr <= stack_virt + page_size - len) {
|
||||
const src: [*]const u8 = @ptrFromInt(ptr);
|
||||
const n = @min(len, write_buf.len - write_len);
|
||||
@memcpy(write_buf[write_len..][0..n], src[0..n]);
|
||||
write_len += n;
|
||||
write_cs = state.cs;
|
||||
log.write("DANOS-INIT: ");
|
||||
log.write(src[0..len]);
|
||||
state.rax = len;
|
||||
} else {
|
||||
state.rax = @bitCast(@as(i64, -1));
|
||||
}
|
||||
},
|
||||
else => state.rax = @bitCast(@as(i64, -1)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Reset the recorded syscall evidence before a user-mode run.
|
||||
fn resetRecords() void {
|
||||
ping_count = 0;
|
||||
write_len = 0;
|
||||
write_cs = 0;
|
||||
exit_code = 0;
|
||||
}
|
||||
|
||||
pub const RunError = error{ ProgramTooBig, OutOfMemory };
|
||||
|
||||
/// Map `blob` at code_virt with a fresh user stack, drop to ring 3, and return
|
||||
/// once the program exits via syscall 0. See the migration caveat in the module
|
||||
/// doc. A program that faults instead never returns (on_fault halts the core).
|
||||
pub fn run(blob: []const u8) RunError!void {
|
||||
if (blob.len > page_size) return error.ProgramTooBig;
|
||||
const code_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const stack_frame = pmm.alloc() orelse {
|
||||
pmm.free(code_frame);
|
||||
return error.OutOfMemory;
|
||||
};
|
||||
|
||||
// Fill the code frame through its identity mapping (supervisor RW): the
|
||||
// user-facing mapping is read-only, and this also sidesteps CR0.WP/SMAP.
|
||||
// The tail is padded with int3 so a stray jump traps instead of sliding.
|
||||
const code: [*]u8 = @ptrFromInt(code_frame);
|
||||
@memcpy(code[0..blob.len], blob);
|
||||
@memset(code[blob.len..page_size], 0xCC);
|
||||
|
||||
arch.setSyscallHandler(syscall);
|
||||
arch.mapUserPage(code_virt, code_frame, false, true); // RO + X
|
||||
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
|
||||
resetRecords();
|
||||
|
||||
arch.enterUser(sched.currentCpuIndex(), code_virt, stack_virt + page_size);
|
||||
|
||||
// Back via the exit syscall; the interrupt gate left IF clear.
|
||||
arch.enableInterrupts();
|
||||
arch.unmapPage(code_virt);
|
||||
arch.unmapPage(stack_virt);
|
||||
pmm.free(code_frame);
|
||||
pmm.free(stack_frame);
|
||||
}
|
||||
|
||||
// --- user ELF loading (/sbin/init) ------------------------------------------
|
||||
|
||||
pub const InitError = error{
|
||||
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
|
||||
BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping
|
||||
BadEntry, // e_entry not inside an executable segment
|
||||
ProgramTooBig, // more pages than the loader's budget
|
||||
OutOfMemory,
|
||||
};
|
||||
|
||||
const max_segments = 16;
|
||||
const max_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||
|
||||
const Segment = struct {
|
||||
vaddr: u64,
|
||||
memsz: u64,
|
||||
filesz: u64,
|
||||
off: u64,
|
||||
writable: bool,
|
||||
executable: bool,
|
||||
|
||||
fn pages(self: Segment) u64 {
|
||||
return (self.memsz + page_size - 1) / page_size;
|
||||
}
|
||||
};
|
||||
|
||||
/// Pages mapped for the running init image, for rollback if loading fails
|
||||
/// midway. On success the mappings stay for the system's life — there's no
|
||||
/// address-space teardown until real processes exist (M3).
|
||||
var loaded = [_]struct { virt: u64, frame: u64 }{.{ .virt = 0, .frame = 0 }} ** (max_pages + 1);
|
||||
var loaded_count: usize = 0;
|
||||
|
||||
/// Parse and validate every PT_LOAD before touching memory. Bounds are checked
|
||||
/// against the image and the user region; segments must be page-aligned,
|
||||
/// non-overlapping, and W^X (R-only is fine — linkers may emit a headers-only
|
||||
/// segment, mapped RO+NX).
|
||||
fn parseSegments(image: []const u8, segs: *[max_segments]Segment) InitError!struct { count: usize, entry: u64 } {
|
||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.BadElf;
|
||||
const ehdr = std.mem.bytesToValue(elf.Elf64_Ehdr, image[0..@sizeOf(elf.Elf64_Ehdr)]);
|
||||
if (!std.mem.eql(u8, image[0..4], "\x7fELF")) return error.BadElf;
|
||||
if (image[elf.EI_CLASS] != elf.ELFCLASS64) return error.BadElf;
|
||||
if (ehdr.e_machine != .X86_64) return error.BadElf;
|
||||
if (ehdr.e_type != .EXEC) return error.BadElf; // a PIE would need relocation
|
||||
if (ehdr.e_phentsize < @sizeOf(elf.Elf64_Phdr)) return error.BadElf;
|
||||
if (ehdr.e_phnum > max_segments) return error.BadElf;
|
||||
const ph_bytes = @as(u64, ehdr.e_phnum) * ehdr.e_phentsize;
|
||||
if (ehdr.e_phoff > image.len or ph_bytes > image.len - ehdr.e_phoff) return error.BadElf;
|
||||
|
||||
var count: usize = 0;
|
||||
var total_pages: u64 = 0;
|
||||
for (0..ehdr.e_phnum) |i| {
|
||||
const off = ehdr.e_phoff + i * ehdr.e_phentsize;
|
||||
const phdr = std.mem.bytesToValue(elf.Elf64_Phdr, image[off..][0..@sizeOf(elf.Elf64_Phdr)]);
|
||||
if (phdr.p_type != elf.PT_LOAD) continue;
|
||||
if (phdr.p_memsz == 0) continue;
|
||||
|
||||
if (phdr.p_vaddr % page_size != 0) return error.BadSegment;
|
||||
if (phdr.p_filesz > phdr.p_memsz) return error.BadSegment;
|
||||
if (phdr.p_offset > image.len or phdr.p_filesz > image.len - phdr.p_offset) return error.BadSegment;
|
||||
// Inside the user image region, strictly below the stack page.
|
||||
if (phdr.p_vaddr < code_virt) return error.BadSegment;
|
||||
if (phdr.p_memsz > stack_virt - phdr.p_vaddr) return error.BadSegment;
|
||||
|
||||
const w = phdr.p_flags & elf.PF_W != 0;
|
||||
const x = phdr.p_flags & elf.PF_X != 0;
|
||||
if (w and x) return error.BadSegment; // W^X, even for init
|
||||
|
||||
const seg = Segment{
|
||||
.vaddr = phdr.p_vaddr,
|
||||
.memsz = phdr.p_memsz,
|
||||
.filesz = phdr.p_filesz,
|
||||
.off = phdr.p_offset,
|
||||
.writable = w,
|
||||
.executable = x,
|
||||
};
|
||||
// No overlap with any earlier segment (page-granular, since mapping is).
|
||||
for (segs[0..count]) |other| {
|
||||
const a_end = seg.vaddr + seg.pages() * page_size;
|
||||
const b_end = other.vaddr + other.pages() * page_size;
|
||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
||||
}
|
||||
total_pages += seg.pages();
|
||||
if (total_pages > max_pages) return error.ProgramTooBig;
|
||||
segs[count] = seg;
|
||||
count += 1;
|
||||
}
|
||||
if (count == 0) return error.BadElf;
|
||||
|
||||
// The entry point must land inside an executable segment.
|
||||
for (segs[0..count]) |seg| {
|
||||
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
|
||||
return .{ .count = count, .entry = ehdr.e_entry };
|
||||
}
|
||||
return error.BadEntry;
|
||||
}
|
||||
|
||||
/// Map one page of a segment: fresh frame, zero + copy through the identity
|
||||
/// mapping (the user mapping may be read-only), then the user-visible mapping
|
||||
/// with the segment's W^X permissions. Records the page for rollback.
|
||||
fn loadPage(image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const dst: [*]u8 = @ptrFromInt(frame);
|
||||
@memset(dst[0..page_size], 0);
|
||||
const page_off = page_index * page_size;
|
||||
if (page_off < seg.filesz) {
|
||||
const n = @min(page_size, seg.filesz - page_off);
|
||||
@memcpy(dst[0..n], image[seg.off + page_off ..][0..n]);
|
||||
}
|
||||
const virt = seg.vaddr + page_off;
|
||||
arch.mapUserPage(virt, frame, seg.writable, seg.executable);
|
||||
loaded[loaded_count] = .{ .virt = virt, .frame = frame };
|
||||
loaded_count += 1;
|
||||
}
|
||||
|
||||
fn unloadAll() void {
|
||||
for (loaded[0..loaded_count]) |p| {
|
||||
arch.unmapPage(p.virt);
|
||||
pmm.free(p.frame);
|
||||
}
|
||||
loaded_count = 0;
|
||||
}
|
||||
|
||||
/// Load a user ELF image, run it in ring 3 from its entry point, and return its
|
||||
/// exit code. Same caller contract as `run` (preemption off, one core). On
|
||||
/// success the user mappings are left in place — teardown comes with real
|
||||
/// processes (M3); on a loading error everything is rolled back.
|
||||
pub fn runInitElf(image: []const u8) InitError!u64 {
|
||||
var segs: [max_segments]Segment = undefined;
|
||||
const parsed = try parseSegments(image, &segs);
|
||||
|
||||
loaded_count = 0;
|
||||
errdefer unloadAll();
|
||||
for (segs[0..parsed.count]) |seg| {
|
||||
for (0..seg.pages()) |i| try loadPage(image, seg, i);
|
||||
}
|
||||
const stack_frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
arch.mapUserPage(stack_virt, stack_frame, true, false); // RW + NX
|
||||
loaded[loaded_count] = .{ .virt = stack_virt, .frame = stack_frame };
|
||||
loaded_count += 1;
|
||||
|
||||
arch.setSyscallHandler(syscall);
|
||||
resetRecords();
|
||||
arch.enterUser(sched.currentCpuIndex(), parsed.entry, stack_virt + page_size);
|
||||
arch.enableInterrupts(); // the exit arrived through an interrupt gate
|
||||
return exit_code;
|
||||
}
|
||||
@@ -109,4 +109,10 @@ pub const BootInfo = extern struct {
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
/// The raw `/sbin/init` ELF image, read off the boot volume by the loader
|
||||
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||
/// kernel boots without user space. Grows into a full initrd handoff later.
|
||||
init_base: u64 = 0,
|
||||
init_len: u64 = 0,
|
||||
};
|
||||
|
||||
@@ -57,6 +57,8 @@ ARCHES = {
|
||||
],
|
||||
"efi_app": ("EFI/BOOT/BOOTX64.efi", "BOOTX64.efi"), # (dest in ESP, name in zig-out/bin)
|
||||
"kernel": ("kernel", "kernel"),
|
||||
# Further files shipped on the ESP: the init user program.
|
||||
"extra": [("sbin/init", "init")],
|
||||
# Built as a function so we can splice in per-run paths.
|
||||
"qemu_args": lambda a, esp, vars_fd, serial: [
|
||||
"-machine", "q35", "-m", "128M",
|
||||
@@ -148,6 +150,21 @@ CASES = [
|
||||
"expect": r"page fault \(vector 14\)",
|
||||
"fail": r"NX not enforced"},
|
||||
{"name": "fault-null", "expect": r"page fault \(vector 14\)"},
|
||||
# Ring 3: a user program runs at CPL 3, makes int 0x80 syscalls, survives
|
||||
# timer interrupts, and exits back into the kernel.
|
||||
{"name": "user",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
|
||||
# 0x5 (present|user) at the user RIP. ([\s\S] spans lines; `.` doesn't.)
|
||||
{"name": "user-pf",
|
||||
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*RIP\s*: 0x00007000000000",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The real user binary: the bootloader ships sbin/init off the ESP, the
|
||||
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
|
||||
{"name": "init",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
|
||||
# pre-transition marker; the FAIL line only appears if the transition didn't take.
|
||||
{"name": "poweroff",
|
||||
@@ -179,6 +196,9 @@ def make_esp(arch):
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(REPO, "zig-out", "bin", efi_name), os.path.join(esp, efi_dest))
|
||||
shutil.copy(os.path.join(REPO, "zig-out", "bin", kern_name), os.path.join(esp, kern_dest))
|
||||
for dest, name in arch.get("extra", []):
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(REPO, "zig-out", "bin", name), os.path.join(esp, dest))
|
||||
return esp
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user