Add process management: enumerate, supervisor-gated kill, exit notifications
process_enumerate snapshots the task table (the device_enumerate shape, so ps is a user program); system_spawn returns the child id, records the caller as supervisor, and takes an exit endpoint; process_kill is allowed only for the supervisor. Every death — exit, fault, or kill — posts a child-exit badge to that endpoint (the IRQ-as-IPC pattern as SIGCHLD). A target caught off-CPU is reaped in place; a running one is condemned and finished at its next system call or tick, guarded so teardown never lands mid-kernel-operation. Tested by process-list, process-kill, and supervision (a ring-3 supervisor exercising the whole surface); design notes in docs/process-management.md.
This commit is contained in:
+177
-34
@@ -129,9 +129,14 @@ pub fn setInitialRamdisk(image: []const u8) void {
|
||||
/// written back into the trap frame, since the entry paths restore user registers
|
||||
/// from it. One handler serves both the system_call/sysret and int-0x80 entry paths.
|
||||
///
|
||||
/// Install it once at boot (before any user code runs) via `init`.
|
||||
/// Install it once at boot (before any user code runs) via `init`. Also registers
|
||||
/// the scheduler's kill hooks: the scheduler sits below this layer, so finishing a
|
||||
/// deferred process_kill (IRQ bindings, IPC handles, the exit notification) is
|
||||
/// called back up into here from the tick (see scheduler.reapKillPendingLocked).
|
||||
pub fn init() void {
|
||||
architecture.setSystemCallHandler(system_call);
|
||||
scheduler.terminate_current_hook = terminateCurrentLocked;
|
||||
scheduler.reap_task_hook = reapTaskLocked;
|
||||
}
|
||||
|
||||
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
|
||||
@@ -140,6 +145,19 @@ fn fail(state: *architecture.CpuState) void {
|
||||
}
|
||||
|
||||
fn system_call(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
const user = t.aspace != 0;
|
||||
if (user) {
|
||||
// A condemned process (process_kill caught it running) dies at its next
|
||||
// kernel entry — before it can spawn, claim, or message anything else.
|
||||
if (t.kill_pending) terminateCurrent();
|
||||
// Mark the span of this call so the timer tick never tears the task down
|
||||
// in the middle of a kernel operation (scheduler.reapKillPendingLocked).
|
||||
t.in_system_call = true;
|
||||
}
|
||||
defer if (user) {
|
||||
t.in_system_call = false;
|
||||
};
|
||||
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
|
||||
.exit => {
|
||||
exit_code = architecture.systemCallArg(state, 0);
|
||||
@@ -178,6 +196,8 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.io_read => systemIoRead(state),
|
||||
.io_write => systemIoWrite(state),
|
||||
.clock => systemClock(state),
|
||||
.process_enumerate => systemProcessEnumerate(state),
|
||||
.process_kill => systemProcessKill(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -415,28 +435,40 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
architecture.setSystemCallResult(state, id);
|
||||
}
|
||||
|
||||
/// system_spawn(name_ptr, name_len, arguments_ptr, arguments_len) -> 0 on success,
|
||||
/// -1 on failure. Load the binary bundled in the initial-ramdisk under `name` as a
|
||||
/// fresh ring-3 process. `name` becomes the child's argv[0] (and its task name, so
|
||||
/// a fault report can say which binary died); `arguments` is an optional
|
||||
/// NUL-separated blob that becomes argv[1..] — how a supervisor parameterises what
|
||||
/// it starts ("you are the driver for device 12"). 0/0 means no extra arguments.
|
||||
/// This is the mechanism a user-space supervisor (the device manager) uses to start
|
||||
/// a driver it matched: discovery and policy stay in user space, the kernel only
|
||||
/// spawns.
|
||||
/// system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint)
|
||||
/// -> the child's process id on success, -1 on failure. Load the binary bundled in
|
||||
/// the initial-ramdisk under `name` as a fresh ring-3 process. `name` becomes the
|
||||
/// child's argv[0] (and its task name, so a fault report can say which binary
|
||||
/// died); `arguments` is an optional NUL-separated blob that becomes argv[1..] —
|
||||
/// how a supervisor parameterises what it starts ("you are the driver for device
|
||||
/// 12"). 0/0 means no extra arguments. This is the mechanism a user-space
|
||||
/// supervisor (the device manager) uses to start a driver it matched: discovery
|
||||
/// and policy stay in user space, the kernel only spawns.
|
||||
///
|
||||
/// Ungated for now — any process may spawn any bundled binary. A capability (only a
|
||||
/// supervisor holds the right to spawn) belongs here once the model grows one; see
|
||||
/// docs/driver-model.md. Both buffers are bounds-checked into the user half exactly
|
||||
/// like `debug_write`, and an unknown name or a load failure returns -1.
|
||||
/// The caller is recorded as the child's **supervisor** — the sole holder of the
|
||||
/// right to `process_kill` it (docs/process-management.md). `exit_endpoint` (a
|
||||
/// handle, or `abi.no_cap` for none) names an endpoint of the caller's to notify
|
||||
/// when the child ends, any way it ends — the IRQ-as-IPC pattern reused as the
|
||||
/// microkernel's SIGCHLD.
|
||||
///
|
||||
/// Spawning itself is still ungated — any process may spawn any bundled binary; a
|
||||
/// spawn capability belongs here once the model grows one (docs/driver-model.md).
|
||||
/// Both buffers are bounds-checked into the user half exactly like `debug_write`,
|
||||
/// and an unknown name or a load failure returns -1.
|
||||
fn systemSpawn(state: *architecture.CpuState) void {
|
||||
const ptr = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
const arguments_ptr = architecture.systemCallArg(state, 2);
|
||||
const arguments_len = architecture.systemCallArg(state, 3);
|
||||
if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||
const exit_handle = architecture.systemCallArg(state, 4);
|
||||
const t = scheduler.current();
|
||||
if (len == 0 or len > scheduler.maximum_task_name or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||
if (arguments_len > maximum_argument_bytes) return fail(state);
|
||||
if (arguments_len != 0 and (arguments_ptr >= user_half_end or arguments_ptr + arguments_len > user_half_end)) return fail(state);
|
||||
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||
null
|
||||
else
|
||||
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||
const image = ramdisk_image orelse return fail(state);
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||
|
||||
@@ -458,19 +490,51 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!std.mem.eql(u8, item.name, name)) continue;
|
||||
spawnProcess(item.blob, 4, argv[0..argc]) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
const child = spawnProcessSupervised(item.blob, 4, argv[0..argc], t.id, exit_endpoint) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, child);
|
||||
return;
|
||||
}
|
||||
fail(state); // no bundled binary by that name
|
||||
}
|
||||
|
||||
/// process_enumerate(buffer, maximum) -> total: snapshot the task table into the
|
||||
/// caller's buffer (up to `maximum` `abi.ProcessDescriptor` entries), returning
|
||||
/// the total live-task count — the exact shape of `device_enumerate`, so a `ps`
|
||||
/// is a user program over a snapshot, not a kernel service. Read-only and
|
||||
/// ungated: what is running is not a secret between cooperating bring-up
|
||||
/// processes.
|
||||
fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(abi.ProcessDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
||||
architecture.setSystemCallResult(state, scheduler.enumerate(out[0..@intCast(cap)]));
|
||||
}
|
||||
|
||||
/// process_kill(id) -> 0 / -ESRCH / -EPERM: end the process `id`. Only its
|
||||
/// supervisor — the process that spawned it — may do so; the supervision link is
|
||||
/// the kill capability, so no user/permission model is needed and a stray id
|
||||
/// cannot be a weapon (ids are never reused, so a stale one just misses).
|
||||
fn systemProcessKill(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = killProcess(t.id, @intCast(id));
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
/// Processes killed by a CPU fault rather than a clean exit. Evidence for the
|
||||
/// fault-recovery test, and a health signal a supervisor can consult later.
|
||||
pub var fault_kill_count: u64 = 0;
|
||||
|
||||
/// Tear down the current user process and reschedule; never returns. Shared by the
|
||||
/// exit system call and the fault path (`killCurrentProcess`). The order matters:
|
||||
/// Release everything a dying task holds and tell its supervisor — the shared
|
||||
/// half of every path out of a process: clean exit, fault kill, and process_kill
|
||||
/// (both the immediate reap and the deferred tick-time terminate). The order
|
||||
/// matters:
|
||||
/// - IRQ bindings are dropped before the handle table closes: dropping the last
|
||||
/// endpoint reference destroys the Endpoint, and a still-bound GSI would have an
|
||||
/// ISR call notifyFromIsr on freed memory the next time the device fired.
|
||||
@@ -479,20 +543,84 @@ pub var fault_kill_count: u64 = 0;
|
||||
/// - A client this task still owes a reply to (it died between receive and reply)
|
||||
/// is failed with -EPEER rather than left blocked forever — a dead server must
|
||||
/// not hang its callers.
|
||||
pub fn terminateCurrent() noreturn {
|
||||
const t = scheduler.current();
|
||||
{
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
irq.releaseOwner(t.id);
|
||||
if (t.ipc_client) |client| {
|
||||
t.ipc_client = null;
|
||||
client.ipc_status = -ipc.EPEER;
|
||||
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||
}
|
||||
ipc.closeHandles(t);
|
||||
/// - The task is unlinked from wherever IPC parked it (an endpoint's sender FIFO,
|
||||
/// a receive wait queue, or a server's owed-reply slot) *before* the handles
|
||||
/// close, so nothing ever dequeues a dangling pointer. These are no-ops for a
|
||||
/// running task ending itself; they matter when process_kill reaps a blocked one.
|
||||
/// - The exit notification is posted last, once the process can no longer act, so
|
||||
/// a supervisor that receives it observes a fully-released child. The endpoint
|
||||
/// reference taken at spawn is dropped with it.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
irq.releaseOwner(t.id);
|
||||
if (t.ipc_client) |client| {
|
||||
t.ipc_client = null;
|
||||
client.ipc_status = -ipc.EPEER;
|
||||
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||
}
|
||||
scheduler.exitUser();
|
||||
ipc.abandonSenderLocked(t);
|
||||
scheduler.removeFromWaitQueueLocked(t);
|
||||
scheduler.forgetIpcClientLocked(t);
|
||||
ipc.closeHandles(t);
|
||||
if (t.exit_endpoint) |raw| {
|
||||
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
|
||||
t.exit_endpoint = null;
|
||||
ipc.notifyLocked(endpoint, abi.notify_exit_bit | t.id);
|
||||
ipc.dropRef(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
/// Tear down the current user process and reschedule; never returns. Shared by the
|
||||
/// exit system call and the fault path (`killCurrentProcess`). See
|
||||
/// `releaseTaskResourcesLocked` for what is released, and in what order.
|
||||
pub fn terminateCurrent() noreturn {
|
||||
_ = sync.enter(); // handed off through the exit switch, released by the resumed task
|
||||
terminateCurrentLocked();
|
||||
}
|
||||
|
||||
/// The body of `terminateCurrent` for a caller that already holds the big kernel
|
||||
/// lock — the scheduler's tick calls this (via `terminate_current_hook`) to finish
|
||||
/// a deferred process_kill on its own core's current task. Never returns; the
|
||||
/// tick's abandoned interrupt frame is fine (the LAPIC was acknowledged before the
|
||||
/// tick hook ran), exactly as on the fault path.
|
||||
fn terminateCurrentLocked() noreturn {
|
||||
releaseTaskResourcesLocked(scheduler.current());
|
||||
scheduler.exitUserLocked();
|
||||
}
|
||||
|
||||
/// Reap a condemned task that is NOT running on any core (ready or blocked — and
|
||||
/// it cannot start running: state changes need the lock we hold). The other half
|
||||
/// of a deferred process_kill, called by the scheduler's tick (via
|
||||
/// `reap_task_hook`) and directly by `killProcess` for targets caught off-CPU.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
fn reapTaskLocked(t: *scheduler.Task) void {
|
||||
releaseTaskResourcesLocked(t);
|
||||
scheduler.removeFromReadyQueueLocked(t); // no-op unless it was ready in a queue
|
||||
scheduler.destroyTaskLocked(t);
|
||||
}
|
||||
|
||||
/// Kill process `target_id` on behalf of `caller_id` — the kernel half of the
|
||||
/// process_kill system call. Returns 0, -ESRCH (no such live process — kernel
|
||||
/// tasks are not killable processes and stale ids miss, since ids are never
|
||||
/// reused), or -EPERM (the caller is not the target's supervisor).
|
||||
///
|
||||
/// A target that is ready or blocked is reaped on the spot. One that is running
|
||||
/// on another core cannot be torn down mid-instruction, so it is condemned
|
||||
/// (`kill_pending`) and dies at its next system_call entry, block, or timer tick
|
||||
/// — like a Unix signal, delivery is prompt but asynchronous. Either way the
|
||||
/// call returns 0: the kill is accepted and irrevocable.
|
||||
pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||
if (target.state == .running) {
|
||||
target.kill_pending = true;
|
||||
} else {
|
||||
reapTaskLocked(target);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Kill the current user process in response to a CPU fault it raised in ring 3.
|
||||
@@ -865,11 +993,21 @@ fn entryStackBytes(argv: []const []const u8) usize {
|
||||
/// convention (`buildEntryStack`). `argv[0]` is required — it names the process:
|
||||
/// the path or initial-ramdisk name it was spawned as. It is also recorded on the
|
||||
/// task, so a fault report can say *which* binary died, not just its id.
|
||||
/// The kernel-internal spawn (init at boot, tests): supervisor 0, no exit
|
||||
/// notification. `spawnProcessSupervised` is the full form.
|
||||
pub fn spawnProcess(image: []const u8, priority: u3, argv: []const []const u8) InitError!void {
|
||||
_ = try spawnProcessSupervised(image, priority, argv, 0, null);
|
||||
}
|
||||
|
||||
/// `spawnProcess`, recording `supervisor` (the id of the process that asked — the
|
||||
/// kill authority) and, if given, `exit_endpoint` to notify when the child ends
|
||||
/// (a reference is taken here and dropped when the notification posts).
|
||||
/// Returns the child's process id.
|
||||
/// Returns immediately — the process runs preemptively on its own page tables
|
||||
/// alongside everything else, and its exit is handled by the system_call layer.
|
||||
/// The whole build (address space + ELF load + task) runs under the kernel lock so
|
||||
/// it appears atomically and can't race pmm/heap on another core.
|
||||
pub fn spawnProcess(image: []const u8, priority: u3, argv: []const []const u8) InitError!void {
|
||||
pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []const u8, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) InitError!u32 {
|
||||
if (argv.len == 0 or argv.len > maximum_arguments) return error.BadArguments;
|
||||
// The entry block must leave most of the page as actual stack.
|
||||
if (entryStackBytes(argv) > page_size / 2) return error.BadArguments;
|
||||
@@ -901,8 +1039,13 @@ pub fn spawnProcess(image: []const u8, priority: u3, argv: []const []const u8) I
|
||||
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
|
||||
}
|
||||
|
||||
if (!scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0]))
|
||||
const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||
return error.OutOfMemory;
|
||||
// The child holds a reference to its exit endpoint from birth to death. Taken
|
||||
// only now, after nothing can fail; the lock is still held, so the child
|
||||
// cannot run (let alone die) before the reference exists.
|
||||
if (exit_endpoint) |endpoint| endpoint.refcount += 1;
|
||||
return child;
|
||||
}
|
||||
|
||||
/// clock() -> nanoseconds since boot: a monotonic time source. The kernel already owns
|
||||
|
||||
Reference in New Issue
Block a user