diff --git a/system/abi.zig b/system/abi.zig index d005fbc..18adda2 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -186,6 +186,7 @@ pub const ProcessState = enum(u32) { pub const ProcessDescriptor = extern struct { id: u32, // kernel-assigned process id; never reused (monotonic) supervisor: u32, // id of the process that spawned it (0 = the kernel) + leader: u32, // process-leader id: == id for a main task, the main task's id for a thread state: u32, // a ProcessState value priority: u32, name_length: u32, diff --git a/system/kernel/process.zig b/system/kernel/process.zig index da1546a..2fb4bba 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -715,17 +715,17 @@ fn systemThreadSpawn(state: *architecture.CpuState) void { null else ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF); - const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state); + const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader) orelse return fail(state); architecture.setSystemCallResult(state, tid); } /// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same** /// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before /// its reference exists. Returns the new thread id, or null on resource exhaustion. -fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 { +fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint, leader: u32) ?u32 { const flags = sync.enter(); defer sync.leave(flags); - const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null; + const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null, leader) orelse return null; if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death return tid; } @@ -972,7 +972,16 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { defer sync.leave(flags); const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH; if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes - if (target.supervisor != caller_id) return -ipc.EPERM; + // The kill authority is the supervision link of the process's LEADER, so the + // capability is per-process, aimed at any member id — a thread's own + // `supervisor` (the task that spawned it) grants nothing here + // (docs/shared-fate-plan.md M1). Kernel tasks were -ESRCH'd above, so a + // leader of 0 is unreachable. + const leader = if (target.leader == target.id) + target + else + scheduler.taskByIdLocked(target.leader) orelse return -ipc.ESRCH; + if (leader.supervisor != caller_id) return -ipc.EPERM; target.exit_reason = .killed; if (target.state == .running) { target.kill_pending = true; @@ -1088,9 +1097,17 @@ fn systemProcessSignal(state: *architecture.CpuState) void { if (signal > 31) return failErr(state, ipc.EBADF); // not a Signal bit position const flags = sync.enter(); defer sync.leave(flags); - const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH); - if (target.address_space == 0) return failErr(state, ipc.ESRCH); - if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM); + const member = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH); + if (member.address_space == 0) return failErr(state, ipc.ESRCH); + // Signals address the process: any member id resolves to the LEADER, whose + // endpoint the service harness binds. Authority mirrors process_kill (the + // leader's supervisor), plus the group may signal itself + // (docs/shared-fate-plan.md M1). + const target = if (member.leader == member.id) + member + else + scheduler.taskByIdLocked(member.leader) orelse return failErr(state, ipc.ESRCH); + if (target.supervisor != t.id and target.id != t.leader) return failErr(state, ipc.EPERM); target.pending_signals |= @as(u32, 1) << @intCast(signal); if (target.signal_endpoint) |raw| { const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw)); @@ -1765,7 +1782,7 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX } - const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse + const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null, 0) orelse return error.OutOfMemory; // The child holds a reference to its exit endpoint from birth to death. Taken // only now, after nothing can fail; the lock is still held, so the child diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 7fef809..738e313 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -49,6 +49,12 @@ pub const Task = struct { // Id of the process that spawned this one (0 = the kernel). The supervision // link is the kill authority: only the supervisor may process_kill a child. supervisor: u32 = 0, + // Id of this task's process leader — the main task's own id, copied to every + // thread it (transitively) spawns; 0 for kernel tasks and never followed. + // Equal `leader` is what makes two tasks one process; the leader's id is the + // id `system_spawn` returned, so it is the process id the supervisor speaks + // (docs/shared-fate-plan.md). + leader: u32 = 0, // Endpoint to notify when this process ends (any way: exit, fault, kill), or // null. Holds its own reference, dropped when the notification is posted. // Opaque here for the same reason as `handles` below. @@ -460,14 +466,16 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { /// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]). /// `supervisor` is the id of the spawning process (0 = the kernel) — the kill /// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller -/// has already taken, or null) is notified when this process ends. +/// has already taken, or null) is notified when this process ends. `leader` is +/// the process leader's id for a thread, or 0 to make the new task its own +/// leader (a process spawn). /// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in /// lands in `user_task_trampoline`. /// Returns the new process id, or null (creating nothing) if the table is full or /// out of memory. /// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it /// across the whole spawn, so the address space and the task appear atomically). -pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 { +pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque, leader: u32) ?u32 { const t = freeSlot() orelse return null; const stack = heap.allocator().alloc(u8, stack_size) catch return null; // Take this task's reference to the address space before we commit the slot, so a @@ -488,6 +496,7 @@ pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u .user_arg = user_arg, .supervisor = supervisor, .exit_endpoint = exit_endpoint, + .leader = if (leader == 0) next_id else leader, }; const name_length = @min(task_name.len, maximum_task_name); @memcpy(t.name_buffer[0..name_length], task_name[0..name_length]); @@ -1035,6 +1044,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 { d.* = .{ .id = t.id, .supervisor = t.supervisor, + .leader = t.leader, .state = @intFromEnum(@as(abi.ProcessState, switch (t.state) { .ready => .ready, .running => .running, diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 50e0648..476ff03 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -1430,7 +1430,7 @@ fn spawnFaultingProcess() ?u32 { architecture.mapUserPageInto(address_space, process.stack_base_virtual, stack_frame, true, false); // RW + NX // Supervised by the calling test task, so exitReasonOf can read the verdict. - const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse { + const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null, 0) orelse { architecture.destroyAddressSpace(address_space); return null; };