From afbf10f7fc4781887b8ecb1937f1b53277d415c4 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Fri, 10 Jul 2026 18:24:46 +0100 Subject: [PATCH] =?UTF-8?q?Device=20manager=20(increment=202):=20system=5F?= =?UTF-8?q?spawn=20=E2=80=94=20actually=20start=20the=20driver?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a `system_spawn(name)` system call: the kernel loads a binary bundled in the initial-ramdisk, by name, as a fresh ring-3 process. It's the mechanism a user-space supervisor needs — discovery and policy stay in user space, the kernel only spawns. The kernel already holds the initial-ramdisk image from the boot handoff; it now stashes it (process.setInitialRamdisk) so the handler can resolve names, bounds-checks the name into the user half like debug_write, and returns -1 for an unknown name or a load failure. Ungated for now (any process may spawn any bundled binary); a spawn capability belongs here once the model grows one. The device manager stops logging "would spawn it" and calls runtime.system.spawn on its matched driver. On QEMU it discovers the HPET, matches `hpet`, and spawns it — and the driver comes all the way up (claims the timer, maps its MMIO, binds and services its IRQ, prints "hpet: ok"). The device-manager test now keys on that final marker: since only the manager is spawned, `hpet: ok` appearing proves the whole discover -> match -> system_spawn -> driver-up chain end to end. Transitional: the kernel still auto-spawns the whole initial-ramdisk at boot, so a real boot briefly double-spawns hpet (the second claim fails harmlessly). Increment 3 removes that redundancy so the manager is the sole owner of driver spawning. Suite 36/36 plus host tests. --- library/runtime/system.zig | 8 ++++ system/abi.zig | 1 + system/kernel/kernel.zig | 2 + system/kernel/process.zig | 41 +++++++++++++++++++ system/kernel/tests.zig | 14 ++++++- .../device-manager/device-manager.zig | 25 ++++++----- 6 files changed, 79 insertions(+), 12 deletions(-) diff --git a/library/runtime/system.zig b/library/runtime/system.zig index 59de2b2..39bb4c4 100644 --- a/library/runtime/system.zig +++ b/library/runtime/system.zig @@ -33,6 +33,14 @@ pub fn exit(code: usize) noreturn { unreachable; // the kernel never returns from exit } +/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 +/// process, returning true on success. This is how a supervisor (the device manager) +/// launches a driver it matched — danos-native, not POSIX (a spawn/exec family comes +/// with the process work later). +pub fn spawn(name: []const u8) bool { + return sc.systemCall2(.system_spawn, @intFromPtr(name.ptr), name.len) == 0; +} + /// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable /// memory and return the base virtual address. On failure returns a value in the /// top page (see `mmapFailed`). The user heap grows through this call. diff --git a/system/abi.zig b/system/abi.zig index 359e6cf..aeb8633 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -36,6 +36,7 @@ pub const SystemCall = enum(u64) { irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed + system_spawn = 17, // system_spawn(name_ptr, name_len) -> 0: start a named initial-ramdisk binary as a new ring-3 process _, }; diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index d7e1f89..7b83e0b 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -304,6 +304,8 @@ fn kmain(boot_information: *const BootInformation) noreturn { fn startInitialRamdiskBinaries(boot_information: *const boot_handoff.BootInformation) void { if (boot_information.initial_ramdisk_len == 0) return; const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + // Hand the image to the process layer so user space can `system_spawn` from it. + process.setInitialRamdisk(image); const rd = initial_ramdisk.Reader.init(image) orelse { status("initial_ramdisk: bad image, skipping\n"); return; diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 4e9a3d6..c5f10ed 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -31,6 +31,7 @@ const sync = @import("sync.zig"); const ipc = @import("ipc-synchronous.zig"); const devices_broker = @import("devices-broker.zig"); const irq = @import("irq.zig"); +const initial_ramdisk = @import("initial-ramdisk"); const log = @import("log.zig"); const page_size = abi.page_size; @@ -81,6 +82,17 @@ pub var write_from_user: bool = false; pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests) pub var exit_code: u64 = 0; +/// The initial-ramdisk image, recorded at boot so `system_spawn` can find bundled +/// binaries by name. Null until `setInitialRamdisk` runs; `system_spawn` then fails +/// cleanly rather than reaching into unset memory. +var ramdisk_image: ?[]const u8 = null; + +/// Record the initial-ramdisk image (the kernel already holds it from the boot +/// handoff) so a user-space supervisor can `system_spawn` binaries out of it. +pub fn setInitialRamdisk(image: []const u8) void { + ramdisk_image = image; +} + /// The system_call surface, dispatched on the saved system_call number (`abi.SystemCall`). /// This is the microkernel-minimal set — memory + scheduling only; file/device /// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is @@ -137,6 +149,7 @@ fn system_call(state: *architecture.CpuState) void { .irq_bind => systemIrqBind(state), .irq_ack => systemIrqAck(state), .device_register => systemDeviceRegister(state), + .system_spawn => systemSpawn(state), _ => fail(state), } } @@ -269,6 +282,34 @@ fn systemDeviceRegister(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, id); } +/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary +/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the +/// mechanism a user-space supervisor (the device manager) uses to start a driver it +/// matched: discovery and policy stay in user space, the kernel only spawns. +/// +/// Ungated for now — any process may spawn any bundled binary. A capability (only a +/// supervisor holds the right to spawn) belongs here once the model grows one; see +/// docs/driver-model.md. The name is bounds-checked into the user half exactly like +/// `debug_write`, and an unknown name or a load failure returns -1. +fn systemSpawn(state: *architecture.CpuState) void { + const ptr = architecture.systemCallArg(state, 0); + const len = architecture.systemCallArg(state, 1); + if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state); + const image = ramdisk_image orelse return fail(state); + const rd = initial_ramdisk.Reader.init(image) orelse return fail(state); + + const name = @as([*]const u8, @ptrFromInt(ptr))[0..len]; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!std.mem.eql(u8, item.name, name)) continue; + spawnProcess(item.blob, 4) catch return fail(state); + architecture.setSystemCallResult(state, 0); + return; + } + fail(state); // no bundled binary by that name +} + /// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed /// (which is what frees the endpoints an ISR would otherwise notify into). fn releaseIrqs(t: *scheduler.Task) void { diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index db23afa..c89a120 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -1196,11 +1196,21 @@ fn deviceManagerTest(boot_information: *const BootInformation) void { return; }; + // Let `system_spawn` find bundled binaries by name (the normal boot path does + // this too). Only the device-manager is spawned here — so if `hpet` runs at all, + // it's because the manager discovered the timer, matched, and spawned it. + process.setInitialRamdisk(image); + process.write_count = 0; process.write_from_user = false; check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager")); - const prefix = "device-manager: ok"; + // End-to-end proof: the driver the manager spawned reaches its own live marker. + // `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO, + // binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after, + // so it's race-free to poll for. Its arrival means the whole + // discover -> match -> system_spawn -> driver-up chain worked. + const prefix = "hpet: ok"; scheduler.setPriority(1); const deadline = architecture.millis() + 10000; while (architecture.millis() < deadline) { @@ -1210,7 +1220,7 @@ fn deviceManagerTest(boot_information: *const BootInformation) void { scheduler.setPriority(4); const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix); - check("device manager enumerated the tree and matched a driver to a device", ok); + check("device manager matched the timer and system_spawn'd hpet, which came up", ok); check("its syscalls came from user mode (CPL 3)", process.write_from_user); result(); } diff --git a/system/services/device-manager/device-manager.zig b/system/services/device-manager/device-manager.zig index 69b0bfd..6671bd6 100644 --- a/system/services/device-manager/device-manager.zig +++ b/system/services/device-manager/device-manager.zig @@ -6,11 +6,12 @@ //! with no special privilege — it uses the same `device_*` system calls any process //! could ([drivers.md](../../../docs/drivers.md), [driver-model.md]). //! -//! Increment 1 (this file): enumerate /system/devices and *match* each device to a -//! driver, logging the decision. It does not spawn anything yet — spawning needs a -//! `system_spawn` system call (the kernel spawns every initial-ramdisk binary in a -//! loop today; see system/kernel/kernel.zig). Increment 2 adds that call and turns -//! these decisions into actual spawns. +//! Increment 2 (this file): enumerate /system/devices, *match* each device to a +//! driver, and *spawn* it with `system_spawn` — the kernel loads the named binary +//! from the initial-ramdisk as a fresh ring-3 process. On QEMU this discovers the +//! HPET, decides `hpet` serves it, and brings that driver all the way up. (The +//! kernel still auto-spawns the whole initial-ramdisk at boot; increment 3 removes +//! that redundancy so the manager is the sole owner of driver spawning.) const runtime = @import("runtime"); const device = runtime.device; @@ -36,12 +37,16 @@ pub fn main() void { var matched: usize = 0; for (buffer[0..count]) |descriptor| { const driver_name = driverFor(descriptor.class) orelse continue; - // Increment 2 will `system_spawn(driver_name)` here; for now, record the - // decision so the policy is observable and testable. - _ = runtime.system.write("device-manager: match "); - _ = runtime.system.write(driver_name); - _ = runtime.system.write(" -> would spawn it\n"); matched += 1; + if (runtime.system.spawn(driver_name)) { + _ = runtime.system.write("device-manager: spawned "); + _ = runtime.system.write(driver_name); + _ = runtime.system.write("\n"); + } else { + _ = runtime.system.write("device-manager: failed to spawn "); + _ = runtime.system.write(driver_name); + _ = runtime.system.write("\n"); + } } if (matched == 0) {