diff --git a/library/runtime/system.zig b/library/runtime/system.zig index 59de2b2..39bb4c4 100644 --- a/library/runtime/system.zig +++ b/library/runtime/system.zig @@ -33,6 +33,14 @@ pub fn exit(code: usize) noreturn { unreachable; // the kernel never returns from exit } +/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 +/// process, returning true on success. This is how a supervisor (the device manager) +/// launches a driver it matched — danos-native, not POSIX (a spawn/exec family comes +/// with the process work later). +pub fn spawn(name: []const u8) bool { + return sc.systemCall2(.system_spawn, @intFromPtr(name.ptr), name.len) == 0; +} + /// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable /// memory and return the base virtual address. On failure returns a value in the /// top page (see `mmapFailed`). The user heap grows through this call. diff --git a/system/abi.zig b/system/abi.zig index 359e6cf..aeb8633 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -36,6 +36,7 @@ pub const SystemCall = enum(u64) { irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed + system_spawn = 17, // system_spawn(name_ptr, name_len) -> 0: start a named initial-ramdisk binary as a new ring-3 process _, }; diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index d7e1f89..7b83e0b 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -304,6 +304,8 @@ fn kmain(boot_information: *const BootInformation) noreturn { fn startInitialRamdiskBinaries(boot_information: *const boot_handoff.BootInformation) void { if (boot_information.initial_ramdisk_len == 0) return; const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + // Hand the image to the process layer so user space can `system_spawn` from it. + process.setInitialRamdisk(image); const rd = initial_ramdisk.Reader.init(image) orelse { status("initial_ramdisk: bad image, skipping\n"); return; diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 4e9a3d6..c5f10ed 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -31,6 +31,7 @@ const sync = @import("sync.zig"); const ipc = @import("ipc-synchronous.zig"); const devices_broker = @import("devices-broker.zig"); const irq = @import("irq.zig"); +const initial_ramdisk = @import("initial-ramdisk"); const log = @import("log.zig"); const page_size = abi.page_size; @@ -81,6 +82,17 @@ pub var write_from_user: bool = false; pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests) pub var exit_code: u64 = 0; +/// The initial-ramdisk image, recorded at boot so `system_spawn` can find bundled +/// binaries by name. Null until `setInitialRamdisk` runs; `system_spawn` then fails +/// cleanly rather than reaching into unset memory. +var ramdisk_image: ?[]const u8 = null; + +/// Record the initial-ramdisk image (the kernel already holds it from the boot +/// handoff) so a user-space supervisor can `system_spawn` binaries out of it. +pub fn setInitialRamdisk(image: []const u8) void { + ramdisk_image = image; +} + /// The system_call surface, dispatched on the saved system_call number (`abi.SystemCall`). /// This is the microkernel-minimal set — memory + scheduling only; file/device /// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is @@ -137,6 +149,7 @@ fn system_call(state: *architecture.CpuState) void { .irq_bind => systemIrqBind(state), .irq_ack => systemIrqAck(state), .device_register => systemDeviceRegister(state), + .system_spawn => systemSpawn(state), _ => fail(state), } } @@ -269,6 +282,34 @@ fn systemDeviceRegister(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, id); } +/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary +/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the +/// mechanism a user-space supervisor (the device manager) uses to start a driver it +/// matched: discovery and policy stay in user space, the kernel only spawns. +/// +/// Ungated for now — any process may spawn any bundled binary. A capability (only a +/// supervisor holds the right to spawn) belongs here once the model grows one; see +/// docs/driver-model.md. The name is bounds-checked into the user half exactly like +/// `debug_write`, and an unknown name or a load failure returns -1. +fn systemSpawn(state: *architecture.CpuState) void { + const ptr = architecture.systemCallArg(state, 0); + const len = architecture.systemCallArg(state, 1); + if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state); + const image = ramdisk_image orelse return fail(state); + const rd = initial_ramdisk.Reader.init(image) orelse return fail(state); + + const name = @as([*]const u8, @ptrFromInt(ptr))[0..len]; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!std.mem.eql(u8, item.name, name)) continue; + spawnProcess(item.blob, 4) catch return fail(state); + architecture.setSystemCallResult(state, 0); + return; + } + fail(state); // no bundled binary by that name +} + /// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed /// (which is what frees the endpoints an ISR would otherwise notify into). fn releaseIrqs(t: *scheduler.Task) void { diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index db23afa..c89a120 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -1196,11 +1196,21 @@ fn deviceManagerTest(boot_information: *const BootInformation) void { return; }; + // Let `system_spawn` find bundled binaries by name (the normal boot path does + // this too). Only the device-manager is spawned here — so if `hpet` runs at all, + // it's because the manager discovered the timer, matched, and spawned it. + process.setInitialRamdisk(image); + process.write_count = 0; process.write_from_user = false; check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager")); - const prefix = "device-manager: ok"; + // End-to-end proof: the driver the manager spawned reaches its own live marker. + // `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO, + // binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after, + // so it's race-free to poll for. Its arrival means the whole + // discover -> match -> system_spawn -> driver-up chain worked. + const prefix = "hpet: ok"; scheduler.setPriority(1); const deadline = architecture.millis() + 10000; while (architecture.millis() < deadline) { @@ -1210,7 +1220,7 @@ fn deviceManagerTest(boot_information: *const BootInformation) void { scheduler.setPriority(4); const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix); - check("device manager enumerated the tree and matched a driver to a device", ok); + check("device manager matched the timer and system_spawn'd hpet, which came up", ok); check("its syscalls came from user mode (CPL 3)", process.write_from_user); result(); } diff --git a/system/services/device-manager/device-manager.zig b/system/services/device-manager/device-manager.zig index 69b0bfd..6671bd6 100644 --- a/system/services/device-manager/device-manager.zig +++ b/system/services/device-manager/device-manager.zig @@ -6,11 +6,12 @@ //! with no special privilege — it uses the same `device_*` system calls any process //! could ([drivers.md](../../../docs/drivers.md), [driver-model.md]). //! -//! Increment 1 (this file): enumerate /system/devices and *match* each device to a -//! driver, logging the decision. It does not spawn anything yet — spawning needs a -//! `system_spawn` system call (the kernel spawns every initial-ramdisk binary in a -//! loop today; see system/kernel/kernel.zig). Increment 2 adds that call and turns -//! these decisions into actual spawns. +//! Increment 2 (this file): enumerate /system/devices, *match* each device to a +//! driver, and *spawn* it with `system_spawn` — the kernel loads the named binary +//! from the initial-ramdisk as a fresh ring-3 process. On QEMU this discovers the +//! HPET, decides `hpet` serves it, and brings that driver all the way up. (The +//! kernel still auto-spawns the whole initial-ramdisk at boot; increment 3 removes +//! that redundancy so the manager is the sole owner of driver spawning.) const runtime = @import("runtime"); const device = runtime.device; @@ -36,12 +37,16 @@ pub fn main() void { var matched: usize = 0; for (buffer[0..count]) |descriptor| { const driver_name = driverFor(descriptor.class) orelse continue; - // Increment 2 will `system_spawn(driver_name)` here; for now, record the - // decision so the policy is observable and testable. - _ = runtime.system.write("device-manager: match "); - _ = runtime.system.write(driver_name); - _ = runtime.system.write(" -> would spawn it\n"); matched += 1; + if (runtime.system.spawn(driver_name)) { + _ = runtime.system.write("device-manager: spawned "); + _ = runtime.system.write(driver_name); + _ = runtime.system.write("\n"); + } else { + _ = runtime.system.write("device-manager: failed to spawn "); + _ = runtime.system.write(driver_name); + _ = runtime.system.write("\n"); + } } if (matched == 0) {