kernel: M1 shared-fate — Task.leader id, kill/signal re-keyed to the leader

Every task carries its process leader's id (main task: own id; threads:
copied from the spawner; kernel tasks: 0, never followed). process_kill and
process_signal resolve any member id to the leader and authorize against the
leader's supervisor, making both capabilities per-process. ProcessDescriptor
gains the leader field. No fan-out yet (docs/shared-fate-plan.md M1).
This commit is contained in:
Daniel Samson
2026-07-22 10:17:13 +01:00
parent e78810195d
commit daca0d9216
4 changed files with 39 additions and 11 deletions
+1
View File
@@ -186,6 +186,7 @@ pub const ProcessState = enum(u32) {
pub const ProcessDescriptor = extern struct { pub const ProcessDescriptor = extern struct {
id: u32, // kernel-assigned process id; never reused (monotonic) id: u32, // kernel-assigned process id; never reused (monotonic)
supervisor: u32, // id of the process that spawned it (0 = the kernel) supervisor: u32, // id of the process that spawned it (0 = the kernel)
leader: u32, // process-leader id: == id for a main task, the main task's id for a thread
state: u32, // a ProcessState value state: u32, // a ProcessState value
priority: u32, priority: u32,
name_length: u32, name_length: u32,
+25 -8
View File
@@ -715,17 +715,17 @@ fn systemThreadSpawn(state: *architecture.CpuState) void {
null null
else else
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF); ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state); const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader) orelse return fail(state);
architecture.setSystemCallResult(state, tid); architecture.setSystemCallResult(state, tid);
} }
/// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same** /// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same**
/// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before /// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before
/// its reference exists. Returns the new thread id, or null on resource exhaustion. /// its reference exists. Returns the new thread id, or null on resource exhaustion.
fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 { fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint, leader: u32) ?u32 {
const flags = sync.enter(); const flags = sync.enter();
defer sync.leave(flags); defer sync.leave(flags);
const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null; const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null, leader) orelse return null;
if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death
return tid; return tid;
} }
@@ -972,7 +972,16 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
defer sync.leave(flags); defer sync.leave(flags);
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH; const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes
if (target.supervisor != caller_id) return -ipc.EPERM; // The kill authority is the supervision link of the process's LEADER, so the
// capability is per-process, aimed at any member id — a thread's own
// `supervisor` (the task that spawned it) grants nothing here
// (docs/shared-fate-plan.md M1). Kernel tasks were -ESRCH'd above, so a
// leader of 0 is unreachable.
const leader = if (target.leader == target.id)
target
else
scheduler.taskByIdLocked(target.leader) orelse return -ipc.ESRCH;
if (leader.supervisor != caller_id) return -ipc.EPERM;
target.exit_reason = .killed; target.exit_reason = .killed;
if (target.state == .running) { if (target.state == .running) {
target.kill_pending = true; target.kill_pending = true;
@@ -1088,9 +1097,17 @@ fn systemProcessSignal(state: *architecture.CpuState) void {
if (signal > 31) return failErr(state, ipc.EBADF); // not a Signal bit position if (signal > 31) return failErr(state, ipc.EBADF); // not a Signal bit position
const flags = sync.enter(); const flags = sync.enter();
defer sync.leave(flags); defer sync.leave(flags);
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH); const member = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
if (target.address_space == 0) return failErr(state, ipc.ESRCH); if (member.address_space == 0) return failErr(state, ipc.ESRCH);
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM); // Signals address the process: any member id resolves to the LEADER, whose
// endpoint the service harness binds. Authority mirrors process_kill (the
// leader's supervisor), plus the group may signal itself
// (docs/shared-fate-plan.md M1).
const target = if (member.leader == member.id)
member
else
scheduler.taskByIdLocked(member.leader) orelse return failErr(state, ipc.ESRCH);
if (target.supervisor != t.id and target.id != t.leader) return failErr(state, ipc.EPERM);
target.pending_signals |= @as(u32, 1) << @intCast(signal); target.pending_signals |= @as(u32, 1) << @intCast(signal);
if (target.signal_endpoint) |raw| { if (target.signal_endpoint) |raw| {
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw)); const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
@@ -1765,7 +1782,7 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX
} }
const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null, 0) orelse
return error.OutOfMemory; return error.OutOfMemory;
// The child holds a reference to its exit endpoint from birth to death. Taken // The child holds a reference to its exit endpoint from birth to death. Taken
// only now, after nothing can fail; the lock is still held, so the child // only now, after nothing can fail; the lock is still held, so the child
+12 -2
View File
@@ -49,6 +49,12 @@ pub const Task = struct {
// Id of the process that spawned this one (0 = the kernel). The supervision // Id of the process that spawned this one (0 = the kernel). The supervision
// link is the kill authority: only the supervisor may process_kill a child. // link is the kill authority: only the supervisor may process_kill a child.
supervisor: u32 = 0, supervisor: u32 = 0,
// Id of this task's process leader — the main task's own id, copied to every
// thread it (transitively) spawns; 0 for kernel tasks and never followed.
// Equal `leader` is what makes two tasks one process; the leader's id is the
// id `system_spawn` returned, so it is the process id the supervisor speaks
// (docs/shared-fate-plan.md).
leader: u32 = 0,
// Endpoint to notify when this process ends (any way: exit, fault, kill), or // Endpoint to notify when this process ends (any way: exit, fault, kill), or
// null. Holds its own reference, dropped when the notification is posted. // null. Holds its own reference, dropped when the notification is posted.
// Opaque here for the same reason as `handles` below. // Opaque here for the same reason as `handles` below.
@@ -460,14 +466,16 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]). /// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
/// `supervisor` is the id of the spawning process (0 = the kernel) — the kill /// `supervisor` is the id of the spawning process (0 = the kernel) — the kill
/// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller /// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller
/// has already taken, or null) is notified when this process ends. /// has already taken, or null) is notified when this process ends. `leader` is
/// the process leader's id for a thread, or 0 to make the new task its own
/// leader (a process spawn).
/// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in /// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in
/// lands in `user_task_trampoline`. /// lands in `user_task_trampoline`.
/// Returns the new process id, or null (creating nothing) if the table is full or /// Returns the new process id, or null (creating nothing) if the table is full or
/// out of memory. /// out of memory.
/// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it /// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it
/// across the whole spawn, so the address space and the task appear atomically). /// across the whole spawn, so the address space and the task appear atomically).
pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 { pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque, leader: u32) ?u32 {
const t = freeSlot() orelse return null; const t = freeSlot() orelse return null;
const stack = heap.allocator().alloc(u8, stack_size) catch return null; const stack = heap.allocator().alloc(u8, stack_size) catch return null;
// Take this task's reference to the address space before we commit the slot, so a // Take this task's reference to the address space before we commit the slot, so a
@@ -488,6 +496,7 @@ pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u
.user_arg = user_arg, .user_arg = user_arg,
.supervisor = supervisor, .supervisor = supervisor,
.exit_endpoint = exit_endpoint, .exit_endpoint = exit_endpoint,
.leader = if (leader == 0) next_id else leader,
}; };
const name_length = @min(task_name.len, maximum_task_name); const name_length = @min(task_name.len, maximum_task_name);
@memcpy(t.name_buffer[0..name_length], task_name[0..name_length]); @memcpy(t.name_buffer[0..name_length], task_name[0..name_length]);
@@ -1035,6 +1044,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
d.* = .{ d.* = .{
.id = t.id, .id = t.id,
.supervisor = t.supervisor, .supervisor = t.supervisor,
.leader = t.leader,
.state = @intFromEnum(@as(abi.ProcessState, switch (t.state) { .state = @intFromEnum(@as(abi.ProcessState, switch (t.state) {
.ready => .ready, .ready => .ready,
.running => .running, .running => .running,
+1 -1
View File
@@ -1430,7 +1430,7 @@ fn spawnFaultingProcess() ?u32 {
architecture.mapUserPageInto(address_space, process.stack_base_virtual, stack_frame, true, false); // RW + NX architecture.mapUserPageInto(address_space, process.stack_base_virtual, stack_frame, true, false); // RW + NX
// Supervised by the calling test task, so exitReasonOf can read the verdict. // Supervised by the calling test task, so exitReasonOf can read the verdict.
const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse { const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null, 0) orelse {
architecture.destroyAddressSpace(address_space); architecture.destroyAddressSpace(address_space);
return null; return null;
}; };