diff --git a/build.zig b/build.zig index 6515810..2438c26 100644 --- a/build.zig +++ b/build.zig @@ -299,6 +299,7 @@ pub fn build(b: *std.Build) void { const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "hpet", "system/drivers/hpet/hpet.zig"); const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "bus", "system/drivers/bus/bus.zig"); const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "device-manager", "system/services/device-manager/device-manager.zig"); + const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, "args-echo", "system/services/args-echo/args-echo.zig"); // Pack the user binaries into the initial_ramdisk image with the host-side Python tool // (the container format is trivial, and Python sidesteps std API churn). Args: @@ -316,6 +317,8 @@ pub fn build(b: *std.Build) void { mk_run.addFileArg(bus_exe.getEmittedBin()); mk_run.addArg("device-manager"); mk_run.addFileArg(device_manager_exe.getEmittedBin()); + mk_run.addArg("args-echo"); + mk_run.addFileArg(args_echo_exe.getEmittedBin()); // Also install the packed binaries to their FHS homes, so zig-out is a true image // of the filesystem — even though at boot they arrive inside the initial-ramdisk. diff --git a/docs/driver-model.md b/docs/driver-model.md index 180d980..6b37e00 100644 --- a/docs/driver-model.md +++ b/docs/driver-model.md @@ -164,8 +164,11 @@ If a class driver needs `mmio`, it has become an HCD and should be one. DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA driver, which is what there is to protect and test against. Proven in the `iommu` test, booted with an emulated `intel-iommu`. -- **`system_spawn`** — a user-space supervisor starts a driver: `system_spawn(name)` - loads a binary bundled in the initial-ramdisk as a fresh ring-3 process. This is what +- **`system_spawn`** — a user-space supervisor starts a driver: + `system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a + fresh ring-3 process; `name` becomes the child's argv[0] and the optional + NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack + ([sysv.md](sysv.md)). This is what turned the device manager from "log the match" into "run the driver": the kernel now spawns only `init`, `init` spawns the services, and the **device-manager** discovers the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a diff --git a/docs/drivers.md b/docs/drivers.md index 699149f..3a1743b 100644 --- a/docs/drivers.md +++ b/docs/drivers.md @@ -37,8 +37,9 @@ kernel ──spawns──► init (PID 1) ──spawns──► device-manag ``` The kernel launches exactly one process — `init` — and hands it nothing but the raw -ability to start more (`system_spawn(name)`, which loads a binary bundled in the -initial-ramdisk as a fresh ring-3 process). Everything else is a user-space decision: +ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled +in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0], +the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision: - **init** ([system/services/init](system/services/init/init.zig)) is the **service supervisor**. It spawns the system services danos brings up at boot — today `vfs` and @@ -52,7 +53,8 @@ initial-ramdisk as a fresh ring-3 process). Everything else is a user-space deci is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads what each driver *binds* (a manifest under `/system/drivers`, or the driver describing its own match). - 3. **Spawn** — `system_spawn(driver_name)` starts the matched driver, which then claims + 3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the + arguments can carry *which* device it matched), which then claims its device and runs the event loop below. So "how is a driver discovered and configured" has two halves: **discovery** is the diff --git a/docs/sysv.md b/docs/sysv.md index e4ed39a..210e51f 100644 --- a/docs/sysv.md +++ b/docs/sysv.md @@ -65,6 +65,36 @@ function-pointer type and the kernel's `_start` both carry whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for the handoff it governs. +## The process-entry stack (argc/argv) + +The SysV ABI also fixes what a *fresh process* finds on its stack — and danos +follows it, so its own runtime and any future C libc read arguments the same way. +At the first user instruction, `rsp` is 16-byte aligned and points at (addresses +growing upward): + +``` +rsp → argc u64 + argv[0] … argv[argc-1] pointers into the strings area below + NULL argv terminator + NULL envp terminator (no environment yet) + {AT_PAGESZ, page size} auxiliary vector + {AT_NULL, 0} auxiliary-vector terminator + argv string bytes NUL-terminated + ───────────────────────── stack top (stack_top_virtual) +``` + +The kernel builds this block at the top of the process's stack — 8 pages (32 KiB, +`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below +them left unmapped as a **guard**, so a stack overflow faults (killing only that +process) instead of silently corrupting the image +(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path +or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional +argument blob becomes `argv[1..]`. The runtime's `_start` +(`library/runtime/start.zig`) hands the block to `rt_start`, which exposes it as +`runtime.argumentCount()` / `runtime.argument(i)`. A C runtime's `crt0` would walk +the identical layout unmodified — that's the compatibility being bought. The +`args` test proves the round trip. + ## Where else it surfaces - **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the diff --git a/library/runtime/runtime.zig b/library/runtime/runtime.zig index e50d9b4..10aba99 100644 --- a/library/runtime/runtime.zig +++ b/library/runtime/runtime.zig @@ -26,5 +26,10 @@ pub const dma = @import("dma.zig"); /// Re-exported so a user binary can `pub const panic = runtime.panic;`. pub const panic = start.panic; +/// Process arguments (argc/argv, parsed from the kernel-built entry stack): +/// `argument(0)` is the path or name this binary was spawned as. +pub const argumentCount = start.argumentCount; +pub const argument = start.argument; + /// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code. pub const allocator = heap.allocator; diff --git a/library/runtime/start.zig b/library/runtime/start.zig index bd6bdfd..2d8afcf 100644 --- a/library/runtime/start.zig +++ b/library/runtime/start.zig @@ -5,25 +5,52 @@ const std = @import("std"); const system = @import("system.zig"); -/// The kernel enters at `_start` with rsp 16-aligned, but a SystemV function expects -/// rsp ≡ 8 (mod 16) on entry (as if reached by `call`). The `call` below pushes -/// the 8-byte return address, satisfying the ABI before any Zig frame runs; the -/// `ud2` is a safety net if `rt_start` ever returns. +/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V +/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the +/// auxiliary vector, then the strings (see system/kernel/process.zig, +/// `buildEntryStack`). Capture that address in rdi — the first SysV argument — +/// before `call` disturbs the stack; the call's pushed return address also puts +/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a +/// safety net if `rt_start` ever returns. pub export fn _start() callconv(.naked) noreturn { asm volatile ( + \\mov %%rsp, %%rdi \\call rt_start \\ud2 ); } -/// The first Zig frame. The heap is lazy (first alloc grows it), so there is no -/// runtime init to order here — just hand control to the program's `main`. -export fn rt_start() callconv(.c) noreturn { +// The process-entry block, recorded by `rt_start` for the accessors below. +var argument_count: usize = 0; +var argument_vector: [*]const u64 = undefined; + +/// The first Zig frame, entered with `stack` pointing at the kernel-built entry +/// block. Record argc/argv for the accessors, then hand control to the program's +/// `main`. The heap is lazy (first alloc grows it), so there is no other runtime +/// init to order here. +export fn rt_start(stack: [*]const u64) callconv(.c) noreturn { + argument_count = stack[0]; + argument_vector = stack + 1; const root = @import("root"); // the user binary's root source file root.main(); system.exit(0); } +/// Number of process arguments (argc). At least 1: argument 0 is the path or +/// name this binary was spawned as. +pub fn argumentCount() usize { + return argument_count; +} + +/// Process argument `index` (0 = the program's own path/name), or an empty slice +/// if out of range. The bytes live in the entry block at the top of the stack +/// page, NUL-terminated, valid for the process's lifetime. +pub fn argument(index: usize) []const u8 { + if (index >= argument_count) return ""; + const string: [*:0]const u8 = @ptrFromInt(argument_vector[index]); + return std.mem.span(string); +} + /// No runtime to unwind into — report a panic as a nonzero exit code. pub const panic = std.debug.FullPanic(struct { fn panic(_: []const u8, _: ?usize) noreturn { diff --git a/library/runtime/system.zig b/library/runtime/system.zig index 40eb40d..4971a1d 100644 --- a/library/runtime/system.zig +++ b/library/runtime/system.zig @@ -44,11 +44,31 @@ pub fn exit(code: usize) noreturn { } /// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 -/// process, returning true on success. This is how a supervisor (the device manager) -/// launches a driver it matched — danos-native, not POSIX (a spawn/exec family comes -/// with the process work later). +/// process, returning true on success. The child's argv[0] is `name`. This is how +/// a supervisor (the device manager) launches a driver it matched — danos-native, +/// not POSIX (a spawn/exec family comes with the process work later). pub fn spawn(name: []const u8) bool { - return sc.systemCall2(.system_spawn, @intFromPtr(name.ptr), name.len) == 0; + return sc.systemCall4(.system_spawn, @intFromPtr(name.ptr), name.len, 0, 0) == 0; +} + +/// Like `spawn`, but hands the child command-line arguments: they arrive as +/// argv[1..] on its System V entry stack (argv[0] is still `name`). Marshalled to +/// the kernel as one NUL-separated blob; the combined arguments must fit +/// `blob` (the kernel caps the blob at 256 bytes and argc at 8 anyway). +pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) bool { + var blob: [256]u8 = undefined; + var len: usize = 0; + for (arguments, 0..) |argument, i| { + if (i != 0) { + if (len >= blob.len) return false; + blob[len] = 0; + len += 1; + } + if (len + argument.len > blob.len) return false; + @memcpy(blob[len..][0..argument.len], argument); + len += argument.len; + } + return sc.systemCall4(.system_spawn, @intFromPtr(name.ptr), name.len, @intFromPtr(&blob), len) == 0; } /// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index ec46be1..c2aa038 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -284,7 +284,7 @@ fn kmain(boot_information: *const BootInformation) noreturn { if (boot_information.init_len != 0) { status("starting /system/services/init...\n"); const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len]; - process.spawnProcess(image, 4) catch |err| { + process.spawnProcess(image, 4, &.{"/system/services/init"}) catch |err| { statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)}); }; } else { @@ -414,7 +414,7 @@ fn recoverableFault(vector: u64) bool { /// is no scheduled process to kill.) fn onException(state: *const architecture.CpuState) noreturn { if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) { - statusPrint("\ndanos: process {d} killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() }); + statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() }); statusPrint(" error code : 0x{x}\n", .{state.error_code}); statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)}); if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address}); diff --git a/system/kernel/process.zig b/system/kernel/process.zig index e664956..73f9ffa 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -24,6 +24,7 @@ const elf = std.elf; const boot_handoff = @import("boot-handoff"); const abi = @import("abi"); const device_abi = @import("device-abi"); +const parameters = @import("parameters"); const architecture = @import("architecture"); const pmm = @import("pmm.zig"); const scheduler = @import("scheduler.zig"); @@ -40,9 +41,18 @@ const SystemCall = abi.SystemCall; /// User virtual addresses. PML4 index 224 — a user-exclusive region, far from /// the identity map (low indices) and the vmm test address (index 128), so /// setting the U/S bit on its intermediate tables widens no kernel mapping. -/// An ELF image may occupy [code_virtual, stack_virtual); the stack page sits above. +/// An ELF image may occupy [code_virtual, stack_virtual); the stack sits above. pub const code_virtual: u64 = 0x0000_7000_0000_0000; -pub const stack_virtual: u64 = 0x0000_7000_0020_0000; + +/// The stack region, above the image. The page at `stack_virtual` is **never +/// mapped** — it is the guard page: a process that overflows its stack walks into +/// it and faults (killing only that process) rather than silently corrupting the +/// top of its own image. The stack proper is `parameters.user_stack_pages` pages +/// at [stack_base_virtual, stack_top_virtual), RW + NX, with the System V entry +/// block (argc/argv) at the very top. +pub const stack_virtual: u64 = 0x0000_7000_0020_0000; // guard page (unmapped) +pub const stack_base_virtual: u64 = stack_virtual + page_size; +pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pages * page_size; /// The mmap grant arena: where `mmap` hands out fresh user pages, above the image /// and stack but still inside PML4[224] (so no kernel mapping is widened). Each @@ -73,6 +83,19 @@ pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per proc /// chunks, so this bound is generous; it also caps the frame scratch array below. const maximum_mmap_pages = 256; +/// Ceiling on a process's argv entries, including argv[0]. Arguments are spawn +/// parameters ("you are the driver for device 12"), not bulk data — IPC carries +/// that — so the bound is small and everything fits the single stack page. +pub const maximum_arguments = 8; + +/// Ceiling on the `system_spawn` extra-arguments blob (argv[1..], NUL-separated). +pub const maximum_argument_bytes = 256; + +/// Auxiliary-vector entry types (System V AMD64 process entry). Only what the +/// kernel emits today; a C runtime scans the vector until the null terminator. +const auxiliary_vector_null: u64 = 0; // AT_NULL — end of the vector +const auxiliary_vector_page_size: u64 = 6; // AT_PAGESZ + // The hand-assembled user program blob (isr.s, .rodata) — the isolation probe. const pf_start = @extern([*]const u8, .{ .name = "user_pf_start" }); const pf_end = @extern([*]const u8, .{ .name = "user_pf_end" }); @@ -392,28 +415,50 @@ fn systemDeviceRegister(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, id); } -/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary -/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the -/// mechanism a user-space supervisor (the device manager) uses to start a driver it -/// matched: discovery and policy stay in user space, the kernel only spawns. +/// system_spawn(name_ptr, name_len, arguments_ptr, arguments_len) -> 0 on success, +/// -1 on failure. Load the binary bundled in the initial-ramdisk under `name` as a +/// fresh ring-3 process. `name` becomes the child's argv[0] (and its task name, so +/// a fault report can say which binary died); `arguments` is an optional +/// NUL-separated blob that becomes argv[1..] — how a supervisor parameterises what +/// it starts ("you are the driver for device 12"). 0/0 means no extra arguments. +/// This is the mechanism a user-space supervisor (the device manager) uses to start +/// a driver it matched: discovery and policy stay in user space, the kernel only +/// spawns. /// /// Ungated for now — any process may spawn any bundled binary. A capability (only a /// supervisor holds the right to spawn) belongs here once the model grows one; see -/// docs/driver-model.md. The name is bounds-checked into the user half exactly like -/// `debug_write`, and an unknown name or a load failure returns -1. +/// docs/driver-model.md. Both buffers are bounds-checked into the user half exactly +/// like `debug_write`, and an unknown name or a load failure returns -1. fn systemSpawn(state: *architecture.CpuState) void { const ptr = architecture.systemCallArg(state, 0); const len = architecture.systemCallArg(state, 1); + const arguments_ptr = architecture.systemCallArg(state, 2); + const arguments_len = architecture.systemCallArg(state, 3); if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state); + if (arguments_len > maximum_argument_bytes) return fail(state); + if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state); const image = ramdisk_image orelse return fail(state); const rd = initial_ramdisk.Reader.init(image) orelse return fail(state); const name = @as([*]const u8, @ptrFromInt(ptr))[0..len]; + var argv: [maximum_arguments][]const u8 = undefined; + argv[0] = name; + var argc: usize = 1; + if (arguments_len != 0) { + const blob = @as([*]const u8, @ptrFromInt(arguments_ptr))[0..arguments_len]; + var pieces = std.mem.tokenizeScalar(u8, blob, 0); + while (pieces.next()) |piece| { + if (argc == maximum_arguments) return fail(state); + argv[argc] = piece; + argc += 1; + } + } + var i: u32 = 0; while (i < rd.count) : (i += 1) { const item = rd.entry(i) orelse continue; if (!std.mem.eql(u8, item.name, name)) continue; - spawnProcess(item.blob, 4) catch return fail(state); + spawnProcess(item.blob, 4, argv[0..argc]) catch return fail(state); architecture.setSystemCallResult(state, 0); return; } @@ -641,15 +686,15 @@ pub fn run(blob: []const u8) RunError!void { @memset(code[blob.len..page_size], 0xCC); architecture.mapUserPage(code_virtual, code_frame, false, true); // RO + X - architecture.mapUserPage(stack_virtual, stack_frame, true, false); // RW + NX + architecture.mapUserPage(stack_base_virtual, stack_frame, true, false); // RW + NX (one page; the probe barely stacks) resetRecords(); - architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_virtual + page_size); + architecture.enterUser(scheduler.currentCpuIndex(), code_virtual, stack_base_virtual + page_size); // Back via the exit system_call; the interrupt gate left IF clear. architecture.enableInterrupts(); architecture.unmapPage(code_virtual); - architecture.unmapPage(stack_virtual); + architecture.unmapPage(stack_base_virtual); pmm.free(code_frame); pmm.free(stack_frame); } @@ -660,6 +705,7 @@ pub const InitError = error{ BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds) BadSegment, // PT_LOAD unaligned, out of the user region, W&X, or overlapping BadEntry, // e_entry not inside an executable segment + BadArguments, // no argv[0], too many entries, or too many bytes for the entry stack ProgramTooBig, // more pages than the loader's budget OutOfMemory, }; @@ -760,13 +806,74 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable); } +/// Build the System V AMD64 process-entry block at the top of a process's stack +/// page and return the initial user stack pointer. At the first user instruction, +/// rsp is 16-byte aligned and points at (addresses growing upward): +/// +/// argc, argv[0..argc-1], NULL, NULL (empty envp), auxiliary vector, strings +/// +/// — the layout every C runtime's startup code walks, so danos's own runtime and a +/// future libc port read arguments identically (docs/sysv.md). `page` is the kernel +/// (physmap) view of the stack's **top** frame and `page_user_base` that frame's +/// user address (stack_top_virtual - page_size); the pointers written into it are +/// user addresses inside that page. The caller has validated the sizes +/// (`entryStackBytes`), so this cannot overrun. +fn buildEntryStack(page: [*]u8, page_user_base: u64, argv: []const []const u8) u64 { + // The strings live at the very top of the page, packed from the end downward. + var string_offset: usize = page_size; + var pointers: [maximum_arguments]u64 = undefined; + var i: usize = argv.len; + while (i > 0) { + i -= 1; + string_offset -= argv[i].len + 1; + @memcpy(page[string_offset..][0..argv[i].len], argv[i]); + page[string_offset + argv[i].len] = 0; // NUL-terminated, as C expects + pointers[i] = page_user_base + string_offset; + } + + // The vector sits below the strings: argc, the argv pointers, the argv + // terminator, an empty envp (terminator only), then the auxiliary vector. + const word_count = 1 + argv.len + 1 + 1 + 4; + const vector_offset = (string_offset - word_count * 8) & ~@as(usize, 15); // entry rsp % 16 == 0 + const words: [*]u64 = @ptrCast(@alignCast(page + vector_offset)); + var w: usize = 0; + words[w] = argv.len; // argc + w += 1; + for (pointers[0..argv.len]) |pointer| { + words[w] = pointer; + w += 1; + } + words[w] = 0; // argv terminator + words[w + 1] = 0; // envp: no environment yet, just the terminator + words[w + 2] = auxiliary_vector_page_size; + words[w + 3] = page_size; + words[w + 4] = auxiliary_vector_null; // end of the auxiliary vector + words[w + 5] = 0; + return page_user_base + vector_offset; +} + +/// Bytes the entry block for `argv` occupies at the top of the stack page: +/// strings (each NUL-terminated), vector words, and the alignment slack. +fn entryStackBytes(argv: []const []const u8) usize { + var string_bytes: usize = 0; + for (argv) |argument| string_bytes += argument.len + 1; + return string_bytes + (1 + argv.len + 1 + 1 + 4) * 8 + 16; +} + /// Load a user ELF image into a fresh address space and spawn it as a scheduled -/// ring-3 process at `priority`. Returns immediately — the process runs -/// preemptively on its own page tables alongside everything else, and its exit -/// is handled by the system_call layer. The whole build (address space + ELF load + -/// task) runs under the kernel lock so it appears atomically and can't race -/// pmm/heap on another core. -pub fn spawnProcess(image: []const u8, priority: u3) InitError!void { +/// ring-3 process at `priority`, entered with `argv` on its stack per the System V +/// convention (`buildEntryStack`). `argv[0]` is required — it names the process: +/// the path or initial-ramdisk name it was spawned as. It is also recorded on the +/// task, so a fault report can say *which* binary died, not just its id. +/// Returns immediately — the process runs preemptively on its own page tables +/// alongside everything else, and its exit is handled by the system_call layer. +/// The whole build (address space + ELF load + task) runs under the kernel lock so +/// it appears atomically and can't race pmm/heap on another core. +pub fn spawnProcess(image: []const u8, priority: u3, argv: []const []const u8) InitError!void { + if (argv.len == 0 or argv.len > maximum_arguments) return error.BadArguments; + // The entry block must leave most of the page as actual stack. + if (entryStackBytes(argv) > page_size / 2) return error.BadArguments; + var segs: [maximum_segments]Segment = undefined; const parsed = try parseSegments(image, &segs); @@ -779,10 +886,22 @@ pub fn spawnProcess(image: []const u8, priority: u3) InitError!void { for (segs[0..parsed.count]) |seg| { for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i); } - const stack_frame = pmm.alloc() orelse return error.OutOfMemory; - architecture.mapUserPageInto(aspace, stack_virtual, stack_frame, true, false); // RW + NX - if (!scheduler.spawnUserLocked(aspace, parsed.entry, stack_virtual + page_size, priority)) + // The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX. + // The page below them (`stack_virtual`) stays unmapped as the overflow guard. + // The entry block goes at the top of the highest page. + var user_sp: u64 = 0; + for (0..parameters.user_stack_pages) |i| { + const stack_frame = pmm.alloc() orelse return error.OutOfMemory; + const stack_page: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(stack_frame)); + @memset(stack_page[0..page_size], 0); // no stale frame contents leak into user space + const page_virtual = stack_base_virtual + i * page_size; + if (i == parameters.user_stack_pages - 1) + user_sp = buildEntryStack(stack_page, page_virtual, argv); + architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX + } + + if (!scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0])) return error.OutOfMemory; } diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index b18b117..97f7cdf 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -70,8 +70,24 @@ pub const Task = struct { ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none) ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none) next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link) + // The process's name — argv[0] as it was spawned (a boot-volume path for init, + // an initial-ramdisk name for everything else); empty for kernel tasks. Fixed + // storage, so the fault path can name the dead without touching the heap. + // Zero-initialised (not `undefined`): an undefined default is materialised as + // a 0xAA fill, which would move the whole static task pool out of .bss. + name_buffer: [maximum_task_name]u8 = .{0} ** maximum_task_name, + name_length: u8 = 0, + + /// The task's name (argv[0] at spawn), or empty for a kernel task. + pub fn name(self: *const Task) []const u8 { + return self.name_buffer[0..self.name_length]; + } }; +/// Capacity of `Task.name_buffer` — matches the longest name `system_spawn` +/// accepts, so a spawned name is never truncated. +pub const maximum_task_name = 64; + /// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because /// it dimensions a field of `Task`; ipc_sync.zig re-exports it. pub const ipc_maximum_handles = 16; @@ -258,12 +274,13 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { } /// Spawn a **user** task: a task with its own address space (`aspace`) that starts -/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for -/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`. +/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]). +/// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in +/// lands in `user_task_trampoline`. /// Returns false (creating nothing) if the table is full or out of memory. /// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it /// across the whole spawn, so the address space and the task appear atomically). -pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool { +pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8) bool { const t = freeSlot() orelse return false; const stack = heap.allocator().alloc(u8, stack_size) catch return false; t.* = .{ @@ -275,6 +292,9 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority .user_ip = entry, .user_sp = user_sp, }; + const name_length = @min(task_name.len, maximum_task_name); + @memcpy(t.name_buffer[0..name_length], task_name[0..name_length]); + t.name_length = @intCast(name_length); next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; t.kstack_top = top; diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 0e3965f..b7c9469 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -120,6 +120,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { userPfTest(); } else if (eql(case, "fault-recovery")) { faultRecoveryTest(boot_information); + } else if (eql(case, "args")) { + argsTest(boot_information); } else if (eql(case, "init")) { initTest(boot_information); } else if (eql(case, "process")) { @@ -1162,8 +1164,8 @@ fn processTest(boot_information: *const BootInformation) void { scheduler.spawn(procWorker, 4); // kernel task at the processes' priority var spawned: u32 = 0; - if (process.spawnProcess(image, 4)) spawned += 1 else |_| {} - if (process.spawnProcess(image, 4)) spawned += 1 else |_| {} + if (process.spawnProcess(image, 4, &.{"/system/services/init"})) spawned += 1 else |_| {} + if (process.spawnProcess(image, 4, &.{"/system/services/init"})) spawned += 1 else |_| {} // Wait (real time) for several heartbeats across the two processes. Each // process sleeps ~1 s between beats, so a few seconds yields several. @@ -1218,9 +1220,9 @@ fn spawnFaultingProcess() bool { architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped return false; }; - architecture.mapUserPageInto(aspace, process.stack_virtual, stack_frame, true, false); // RW + NX + architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX - if (!scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_virtual + abi.page_size, 4)) { + if (!scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe")) { architecture.destroyAddressSpace(aspace); return false; } @@ -1243,7 +1245,7 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void { process.write_count = 0; process.fault_kill_count = 0; - const spawned = if (process.spawnProcess(image, 4)) true else |_| false; + const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |_| false; check("init spawned as the surviving process", spawned); // A first heartbeat proves init runs before the fault. @@ -1287,7 +1289,7 @@ fn initTest(boot_information: *const BootInformation) void { } const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len]; process.write_count = 0; - const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: { + const spawned = if (process.spawnProcess(image, 4, &.{"/system/services/init"})) true else |err| blk: { log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)}); break :blk false; }; @@ -1334,7 +1336,7 @@ fn initialRamdiskTest(boot_information: *const BootInformation) void { var i: u32 = 0; while (i < rd.count) : (i += 1) { const item = rd.entry(i) orelse continue; - if (process.spawnProcess(item.blob, 4)) spawned += 1 else |err| { + if (process.spawnProcess(item.blob, 4, &.{item.name})) spawned += 1 else |err| { log("DANOS-INITRD-ERR: {s}: {s}\n", .{ item.name, @errorName(err) }); } } @@ -1394,6 +1396,44 @@ fn vfsTest(boot_information: *const BootInformation) void { result(); } +/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the +/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through +/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument +/// blob. Instance 2 parses the kernel-built System V entry stack via the runtime +/// and echoes its whole argv in one write, which must arrive exactly as sent. +fn argsTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: args\n", .{}); + check("bootloader handed over an initial_ramdisk", boot_information.initial_ramdisk_len != 0); + if (boot_information.initial_ramdisk_len == 0) { + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + process.setInitialRamdisk(image); // args-echo respawns itself through system_spawn + + process.write_count = 0; + process.write_from_user = false; + check("args-echo spawned from the initial_ramdisk", spawnNamed(rd, "args-echo")); + + // Wait for the *second* instance's echo (the first writes nothing). + scheduler.setPriority(1); + const deadline = architecture.millis() + 8000; + while (process.write_count < 1 and architecture.millis() < deadline) scheduler.yield(); + scheduler.setPriority(4); + + const expected = "args: args-echo alpha beta-42\n"; + const echoed = process.write_len == expected.len and eql(process.write_buffer[0..process.write_len], expected); + if (!echoed and process.write_len > 0) log("DANOS-ARGS: got \"{s}\"\n", .{process.write_buffer[0..process.write_len]}); + check("argv arrived intact (argv[0] = name, argv[1..] = spawn arguments)", echoed); + check("echo came from user mode (CPL 3)", process.write_from_user); + result(); +} + /// Spawn the initial_ramdisk binary named `name` as a ring-3 process. Returns false if it /// isn't in the image or fails to load. fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool { @@ -1401,7 +1441,7 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool { while (i < rd.count) : (i += 1) { const item = rd.entry(i) orelse continue; if (eql(item.name, name)) { - return if (process.spawnProcess(item.blob, 4)) true else |_| false; + return if (process.spawnProcess(item.blob, 4, &.{item.name})) true else |_| false; } } return false; diff --git a/system/parameters.zig b/system/parameters.zig index baf041c..2f01b31 100644 --- a/system/parameters.zig +++ b/system/parameters.zig @@ -22,6 +22,12 @@ pub const maximum_tasks = 16; /// Each task's kernel stack (also each AP's bring-up stack), in bytes. pub const kernel_stack_size = 16 * 1024; +/// Each user process's stack, in pages (32 KiB). Mapped just below a fixed top; +/// the System V entry block (argc/argv) occupies the top of the highest page, and +/// the page below the mapping is left unmapped as a guard, so an overflow faults +/// (killing only that process) instead of silently corrupting the image. +pub const user_stack_pages = 8; + /// Each core's IST (double-fault) stack, in bytes. The BSP's is static; an AP's is /// heap-allocated at bring-up. pub const ist_stack_size = 16 * 1024; diff --git a/system/services/args-echo/args-echo.zig b/system/services/args-echo/args-echo.zig new file mode 100644 index 0000000..8ae582e --- /dev/null +++ b/system/services/args-echo/args-echo.zig @@ -0,0 +1,55 @@ +//! args-echo — a test fixture for process arguments (bundled in the +//! initial-ramdisk, spawned only by the `args` test case). Run with no arguments, +//! it respawns itself *with* some via `spawnWithArguments` — exercising the +//! system_spawn argument blob. Run with arguments, it burns more stack than one +//! page could hold (proving the multi-page stack: on a single-page stack the +//! recursion would hit the guard and the process would be killed before echoing), +//! then echoes its whole argv in one `debug_write` the kernel test asserts on — +//! proving the kernel-built System V entry stack (argc, argv pointers, +//! NUL-terminated strings) and the runtime's parsing of it, end to end. + +const runtime = @import("runtime"); + +/// Recurse with a real frame each level: `depth` levels of ~0.5 KiB, touched +/// through a volatile pointer so no optimiser can flatten the frames away. +fn burnStack(depth: usize) u8 { + var frame: [512]u8 = undefined; + const touch: *volatile [512]u8 = &frame; + touch[0] = @truncate(depth); + touch[511] = touch[0]; + if (depth == 0) return touch[511]; + return touch[0] +% burnStack(depth - 1); +} + +pub fn main() void { + if (runtime.argumentCount() <= 1) { + // First instance: spawn the second with real arguments, then exit. + _ = runtime.system.spawnWithArguments("args-echo", &.{ "alpha", "beta-42" }); + return; + } + + // ~16 x 0.5 KiB frames: comfortably past one page, well inside the 32 KiB stack. + _ = burnStack(16); + + // Second instance: echo "args: ..." for the test to match. + var buffer: [128]u8 = undefined; + const prefix = "args:"; + @memcpy(buffer[0..prefix.len], prefix); + var len: usize = prefix.len; + for (0..runtime.argumentCount()) |i| { + const argument = runtime.argument(i); + if (len + 1 + argument.len + 1 > buffer.len) break; + buffer[len] = ' '; + len += 1; + @memcpy(buffer[len..][0..argument.len], argument); + len += argument.len; + } + buffer[len] = '\n'; + len += 1; + _ = runtime.system.write(buffer[0..len]); +} + +pub const panic = runtime.panic; +comptime { + _ = &runtime.start._start; // pull the runtime entry shim into the image +} diff --git a/test/qemu_test.py b/test/qemu_test.py index 2ff505a..efd33e8 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -205,6 +205,11 @@ CASES = [ "timeout": 60, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned + # name, argv[1..] = the system_spawn argument blob) and echoes back intact. + {"name": "args", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # The real user binary: the bootloader ships /system/services/init off the ESP, the # kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly. {"name": "init",