Add process management: enumerate, supervisor-gated kill, exit notifications
process_enumerate snapshots the task table (the device_enumerate shape, so ps is a user program); system_spawn returns the child id, records the caller as supervisor, and takes an exit endpoint; process_kill is allowed only for the supervisor. Every death — exit, fault, or kill — posts a child-exit badge to that endpoint (the IRQ-as-IPC pattern as SIGCHLD). A target caught off-CPU is reaped in place; a running one is condemned and finished at its next system call or tick, guarded so teardown never lands mid-kernel-operation. Tested by process-list, process-kill, and supervision (a ring-3 supervisor exercising the whole surface); design notes in docs/process-management.md.
This commit is contained in:
+206
-11
@@ -18,6 +18,7 @@
|
||||
//! shared queues.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const heap = @import("heap.zig");
|
||||
@@ -41,6 +42,26 @@ pub const Task = struct {
|
||||
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
|
||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
||||
// --- process management (process.zig) ---
|
||||
// Id of the process that spawned this one (0 = the kernel). The supervision
|
||||
// link is the kill authority: only the supervisor may process_kill a child.
|
||||
supervisor: u32 = 0,
|
||||
// Endpoint to notify when this process ends (any way: exit, fault, kill), or
|
||||
// null. Holds its own reference, dropped when the notification is posted.
|
||||
// Opaque here for the same reason as `handles` below.
|
||||
exit_endpoint: ?*anyopaque = null,
|
||||
// Set by process_kill on a task that is running on another core; the kernel
|
||||
// finishes the kill at that task's next system call or timer tick.
|
||||
kill_pending: bool = false,
|
||||
// True while this task executes its own system call — the timer tick must not
|
||||
// tear a task down in the middle of a kernel operation, only while it runs
|
||||
// user code (or sits at a block point, where teardown is safe).
|
||||
in_system_call: bool = false,
|
||||
// Where this task is parked while blocked, so a kill can unlink it: the
|
||||
// WaitQueue it waits on (maintained by waitLocked/wakeLocked), or the endpoint
|
||||
// whose sender FIFO it queues in (maintained by the IPC layer; opaque here).
|
||||
wait_queue: ?*WaitQueue = null,
|
||||
ipc_wait_endpoint: ?*anyopaque = null,
|
||||
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||
// runs on the shared kernel page tables). A user task carries its own.
|
||||
aspace: u64 = 0,
|
||||
@@ -85,8 +106,9 @@ pub const Task = struct {
|
||||
};
|
||||
|
||||
/// Capacity of `Task.name_buffer` — matches the longest name `system_spawn`
|
||||
/// accepts, so a spawned name is never truncated.
|
||||
pub const maximum_task_name = 64;
|
||||
/// accepts, so a spawned name is never truncated. Shared with the ABI's
|
||||
/// ProcessDescriptor, so `enumerate` copies names without clipping.
|
||||
pub const maximum_task_name = abi.maximum_process_name;
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
@@ -275,14 +297,18 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
|
||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
||||
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
|
||||
/// `supervisor` is the id of the spawning process (0 = the kernel) — the kill
|
||||
/// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller
|
||||
/// has already taken, or null) is notified when this process ends.
|
||||
/// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in
|
||||
/// lands in `user_task_trampoline`.
|
||||
/// Returns false (creating nothing) if the table is full or out of memory.
|
||||
/// Returns the new process id, or null (creating nothing) if the table is full or
|
||||
/// out of memory.
|
||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
||||
/// across the whole spawn, so the address space and the task appear atomically).
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8) bool {
|
||||
const t = freeSlot() orelse return false;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||
const t = freeSlot() orelse return null;
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch return null;
|
||||
t.* = .{
|
||||
.id = next_id,
|
||||
.state = .ready,
|
||||
@@ -291,6 +317,8 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
||||
.aspace = aspace,
|
||||
.user_ip = entry,
|
||||
.user_sp = user_sp,
|
||||
.supervisor = supervisor,
|
||||
.exit_endpoint = exit_endpoint,
|
||||
};
|
||||
const name_length = @min(task_name.len, maximum_task_name);
|
||||
@memcpy(t.name_buffer[0..name_length], task_name[0..name_length]);
|
||||
@@ -302,7 +330,7 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority
|
||||
// the user entry/stack from the Task itself).
|
||||
t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask));
|
||||
enqueue(t);
|
||||
return true;
|
||||
return t.id;
|
||||
}
|
||||
|
||||
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
|
||||
@@ -414,6 +442,7 @@ pub const WaitQueue = struct {
|
||||
pub fn waitLocked(wait_queue: *WaitQueue) void {
|
||||
const t = current();
|
||||
t.state = .blocked;
|
||||
t.wait_queue = wait_queue; // so a kill can unlink a parked waiter
|
||||
t.next = wait_queue.head;
|
||||
wait_queue.head = t;
|
||||
schedule();
|
||||
@@ -438,10 +467,78 @@ pub fn wakeLocked(wait_queue: *WaitQueue) void {
|
||||
}
|
||||
const t = best orelse return;
|
||||
if (best_previous) |p| p.next = t.next else wait_queue.head = t.next;
|
||||
t.wait_queue = null;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Unlink `t` from the wait queue it is parked on, if any (the kill path — a
|
||||
/// killed waiter must not be woken later as a dangling pointer). Precondition:
|
||||
/// the big kernel lock is held.
|
||||
pub fn removeFromWaitQueueLocked(t: *Task) void {
|
||||
const wait_queue = t.wait_queue orelse return;
|
||||
t.wait_queue = null;
|
||||
var previous: ?*Task = null;
|
||||
var node = wait_queue.head;
|
||||
while (node) |n| : ({
|
||||
previous = n;
|
||||
node = n.next;
|
||||
}) {
|
||||
if (n != t) continue;
|
||||
if (previous) |p| p.next = t.next else wait_queue.head = t.next;
|
||||
t.next = null;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
/// Unlink `t` from the ready queue it sits in (global, or its affinity core's
|
||||
/// pinned queue) — the kill path for a task that is runnable but not running.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn removeFromReadyQueueLocked(t: *Task) void {
|
||||
if (t.affinity) |cpu| {
|
||||
const pc = &cpus[cpu];
|
||||
removeFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
|
||||
} else {
|
||||
removeFrom(&ready_head, &ready_tail, &ready_bitmap, t);
|
||||
}
|
||||
}
|
||||
|
||||
fn removeFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
||||
const level: usize = t.priority;
|
||||
var previous: ?*Task = null;
|
||||
var node = head[level];
|
||||
while (node) |n| : ({
|
||||
previous = n;
|
||||
node = n.next;
|
||||
}) {
|
||||
if (n != t) continue;
|
||||
if (previous) |p| p.next = t.next else head[level] = t.next;
|
||||
if (tail[level] == t) tail[level] = previous;
|
||||
if (head[level] == null) bitmap.* &= ~(@as(u8, 1) << @intCast(level));
|
||||
t.next = null;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
/// Find a live task by process id, or null. Ids are monotonic and never reused,
|
||||
/// so a stale id misses cleanly rather than naming a recycled slot.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn taskByIdLocked(id: u32) ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state != .free and t.id == id) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Make every server that still holds `t` as the client it owes a reply to forget
|
||||
/// it — the reply of a dead client is dropped, not delivered into freed state.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
pub fn forgetIpcClientLocked(t: *Task) void {
|
||||
for (&tasks) |*other| {
|
||||
if (other.state != .free and other.ipc_client == t) other.ipc_client = null;
|
||||
}
|
||||
}
|
||||
|
||||
/// Block the current task and switch away, without putting it on any wait queue —
|
||||
/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender
|
||||
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
||||
@@ -501,14 +598,55 @@ fn wakeExpired() void {
|
||||
}
|
||||
}
|
||||
|
||||
// Process-teardown hooks, registered by process.zig at init — the scheduler sits
|
||||
// below the process layer, so finishing a kill (IRQ bindings, IPC handles, exit
|
||||
// notification) is called *up* through these, mirroring how the architecture
|
||||
// layer calls up into `tick`.
|
||||
//
|
||||
// `terminate_current_hook` ends the task running on THIS core (lock held, never
|
||||
// returns — it switches away like `exitUserLocked`). `reap_task_hook` tears down
|
||||
// a task that is NOT running on any core (lock held).
|
||||
pub var terminate_current_hook: ?*const fn () noreturn = null;
|
||||
pub var reap_task_hook: ?*const fn (*Task) void = null;
|
||||
|
||||
/// Finish any pending kills this core can see (the deferred half of process_kill;
|
||||
/// the immediate half runs in the killer's own call). Precondition: the big kernel
|
||||
/// lock is held, from `tick`.
|
||||
///
|
||||
/// - This core's *current* task, if condemned, is terminated here — but only when
|
||||
/// it is not inside one of its own system calls (`in_system_call`): the tick may
|
||||
/// have interrupted kernel code mid-operation, where teardown would leak or
|
||||
/// corrupt what that operation holds. User-mode execution (and the system_call
|
||||
/// entry/exit stubs, which hold nothing) are safe termination points. A task
|
||||
/// that *is* mid-call dies at its next block, tick, or system_call entry instead.
|
||||
/// The hook never returns; abandoning the interrupt frame is fine — the LAPIC
|
||||
/// was acknowledged before the tick hook ran (see apic.timerTick), exactly as on
|
||||
/// the fault-kill path.
|
||||
/// - Condemned tasks that are ready or blocked are not running anywhere (state
|
||||
/// changes need the lock we hold), so they are reaped in place.
|
||||
fn reapKillPendingLocked() void {
|
||||
const pc = thisCpu();
|
||||
const cur = pc.current;
|
||||
if (cur.kill_pending and cur.aspace != 0 and !cur.in_system_call) {
|
||||
if (terminate_current_hook) |hook| hook(); // noreturn
|
||||
}
|
||||
if (reap_task_hook) |hook| {
|
||||
for (&tasks) |*t| {
|
||||
if (!t.kill_pending) continue;
|
||||
if (t.state == .ready or t.state == .blocked) hook(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
||||
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
|
||||
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
|
||||
/// restores the interrupted context's flags, so re-enabling here would open a
|
||||
/// nested-interrupt window before the return.
|
||||
/// sleepers, finish pending kills, then preempt. Takes the kernel lock like any
|
||||
/// other critical section, but releases it *without* touching the interrupt flag
|
||||
/// — the handler's `iretq` restores the interrupted context's flags, so
|
||||
/// re-enabling here would open a nested-interrupt window before the return.
|
||||
pub fn tick() void {
|
||||
_ = sync.enter();
|
||||
wakeExpired();
|
||||
reapKillPendingLocked();
|
||||
if (preemption_enabled) schedule();
|
||||
sync.leaveIsr();
|
||||
}
|
||||
@@ -540,6 +678,13 @@ pub fn exit() noreturn {
|
||||
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
|
||||
pub fn exitUser() noreturn {
|
||||
_ = sync.enter();
|
||||
exitUserLocked();
|
||||
}
|
||||
|
||||
/// The body of `exitUser` for callers that already hold the big kernel lock (the
|
||||
/// tick-time terminate path, which enters with the lock held). The lock is handed
|
||||
/// off through the switch and released by the task that resumes. Never returns.
|
||||
pub fn exitUserLocked() noreturn {
|
||||
const pc = thisCpu();
|
||||
const dying = pc.current;
|
||||
const as = dying.aspace;
|
||||
@@ -551,6 +696,8 @@ pub fn exitUser() noreturn {
|
||||
}
|
||||
dying.state = .free;
|
||||
dying.aspace = 0;
|
||||
dying.kill_pending = false;
|
||||
dying.in_system_call = false;
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
@@ -559,6 +706,54 @@ pub fn exitUser() noreturn {
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Free a task that is NOT running on any core (it is ready or blocked, and the
|
||||
/// caller — the kill path — has already unlinked it from every queue and released
|
||||
/// what it held). Destroys its address space: safe here because no core can have
|
||||
/// it loaded (every switch away from a task loads the next task's tables, and the
|
||||
/// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper
|
||||
/// yet). Precondition: the big kernel lock is held.
|
||||
pub fn destroyTaskLocked(t: *Task) void {
|
||||
if (t.aspace != 0) architecture.destroyAddressSpace(t.aspace);
|
||||
t.aspace = 0;
|
||||
t.kill_pending = false;
|
||||
t.in_system_call = false;
|
||||
t.wake_at = 0;
|
||||
t.state = .free;
|
||||
}
|
||||
|
||||
/// Snapshot the task table into `out` (up to its length), returning the total
|
||||
/// number of live tasks — the kernel half of `process_enumerate`, mirroring
|
||||
/// devices_broker.enumerate. Kernel tasks are included (empty name, supervisor 0):
|
||||
/// an honest `ps` shows the idle tasks too. `out` may be user memory: the caller's
|
||||
/// address space is loaded during its system call, and the same bring-up trust
|
||||
/// applies as for device_enumerate (an unmapped user page faults the kernel).
|
||||
pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
var total: u64 = 0;
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) continue;
|
||||
if (total < out.len) {
|
||||
const d = &out[total];
|
||||
d.* = .{
|
||||
.id = t.id,
|
||||
.supervisor = t.supervisor,
|
||||
.state = @intFromEnum(@as(abi.ProcessState, switch (t.state) {
|
||||
.ready => .ready,
|
||||
.running => .running,
|
||||
.blocked => .blocked,
|
||||
.free => unreachable,
|
||||
})),
|
||||
.priority = t.priority,
|
||||
.name_length = t.name_length,
|
||||
.name = t.name_buffer,
|
||||
};
|
||||
}
|
||||
total += 1;
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/// Whether the running task is a user process (has its own address space).
|
||||
pub fn currentIsUserProcess() bool {
|
||||
return current().aspace != 0;
|
||||
|
||||
Reference in New Issue
Block a user