Signals over IPC, one-shot timers, and the service harness (M17.4)

Signals are statements delivered as coalescing notifications to the
endpoint a process nominates with signal_bind — never a hijacked stack,
never a question (liveness is the zero-length ping the harness answers).
process_signal is supervisor-or-self gated, like kill; unbound targets
accumulate a pending mask delivered on bind. timer_bind is the missing
timed wait: a one-shot deadline landing in the same replyWait as
everything else — what stop(), hello deadlines, and restart backoff are
built from. runtime.service.run folds requests, signals, and
notifications into callbacks; the VFS conversion deletes its hand-rolled
loop and gains the whole lifecycle contract. The signals scenario drives
ping, reload, terminate->exited, the timer, and the deaf-child
deadline->killed path from ring 3. docs/process-lifecycle.md increments
1-4 are now as-built.
This commit is contained in:
Daniel Samson
2026-07-12 23:53:38 +01:00
parent d8c55c6f2f
commit 650a1b1595
15 changed files with 507 additions and 30 deletions
+104
View File
@@ -137,6 +137,7 @@ pub fn init() void {
architecture.setSystemCallHandler(system_call);
scheduler.terminate_current_hook = terminateCurrentLocked;
scheduler.reap_task_hook = reapTaskLocked;
scheduler.timer_tick_hook = timerSweepLocked;
}
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
@@ -202,6 +203,9 @@ fn system_call(state: *architecture.CpuState) void {
.process_kill => systemProcessKill(state),
.process_exit_reason => systemProcessExitReason(state),
.process_subscribe => systemProcessSubscribe(state),
.signal_bind => systemSignalBind(state),
.process_signal => systemProcessSignal(state),
.timer_bind => systemTimerBind(state),
_ => fail(state),
}
}
@@ -575,6 +579,20 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
recordExitLocked(t);
irq.releaseOwner(t.id);
devices_broker.releaseAllOwnedBy(t.id);
// The dying task's signal endpoint and one-shot timers go with it.
if (t.signal_endpoint) |raw| {
ipc.dropRef(@ptrCast(@alignCast(raw)));
t.signal_endpoint = null;
}
t.pending_signals = 0;
for (&one_shot_timers) |*slot| {
if (slot.*) |timer| {
if (timer.owner == t.id) {
ipc.dropRef(timer.endpoint);
slot.* = null;
}
}
}
// A dead subscriber's own subscriptions go first: it must not hear about
// itself, and the slots' endpoint references drop with it.
for (&exit_subscribers) |*slot| {
@@ -735,6 +753,92 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
failErr(state, ipc.ENOSPC);
}
/// signal_bind(endpoint): nominate where this process's signals arrive — the
/// IRQ-as-IPC pattern a fourth time (docs/process-lifecycle.md). Replacing a
/// binding drops the old reference; signals that pended while unbound are
/// delivered immediately on bind, coalesced into one notification.
fn systemSignalBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
const flags = sync.enter();
defer sync.leave(flags);
if (t.signal_endpoint) |raw| ipc.dropRef(@ptrCast(@alignCast(raw)));
endpoint.refcount += 1;
t.signal_endpoint = @ptrCast(endpoint);
if (t.pending_signals != 0) {
ipc.notifyLocked(endpoint, abi.notify_signal_bit | t.pending_signals);
t.pending_signals = 0;
}
architecture.setSystemCallResult(state, 0);
}
/// process_signal(id, signal): post a signal — a one-way, coalescing statement,
/// never a question (docs/process-lifecycle.md). The authority gate is the
/// supervision link, like kill; a process may also signal itself. Unbound
/// targets accumulate the signal in their pending mask.
fn systemProcessSignal(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const id = architecture.systemCallArg(state, 0);
const signal = architecture.systemCallArg(state, 1);
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
if (signal > 31) return failErr(state, ipc.EBADF); // not a Signal bit position
const flags = sync.enter();
defer sync.leave(flags);
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
if (target.aspace == 0) return failErr(state, ipc.ESRCH);
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM);
target.pending_signals |= @as(u32, 1) << @intCast(signal);
if (target.signal_endpoint) |raw| {
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
ipc.notifyLocked(endpoint, abi.notify_signal_bit | target.pending_signals);
target.pending_signals = 0;
}
architecture.setSystemCallResult(state, 0);
}
/// The one-shot timers of timer_bind: the missing timed wait. A service arms a
/// deadline and keeps serving; the expiry arrives in the same replyWait as
/// everything else (notify_timer_bit). What stop-sequence escalation, hello
/// deadlines, and restart backoff are built from — and later, `alarm`.
const timer_capacity = 16;
const OneShotTimer = struct { deadline: u64, endpoint: *ipc.Endpoint, owner: u32 };
var one_shot_timers: [timer_capacity]?OneShotTimer = .{null} ** timer_capacity;
/// Sweep expired timers — hung on scheduler.timer_tick_hook, so it runs on every
/// tick with the big kernel lock held, like the sleeper wake it rides beside.
fn timerSweepLocked() void {
const now = architecture.millis();
for (&one_shot_timers) |*slot| {
if (slot.*) |timer| {
if (now >= timer.deadline) {
ipc.notifyLocked(timer.endpoint, abi.notify_timer_bit);
ipc.dropRef(timer.endpoint);
slot.* = null;
}
}
}
}
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
fn systemTimerBind(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
const ms = architecture.systemCallArg(state, 1);
const flags = sync.enter();
defer sync.leave(flags);
for (&one_shot_timers) |*slot| {
if (slot.* == null) {
endpoint.refcount += 1;
slot.* = .{ .deadline = architecture.millis() + ms, .endpoint = endpoint, .owner = t.id };
return architecture.setSystemCallResult(state, 0);
}
}
failErr(state, ipc.ENOSPC);
}
fn systemProcessExitReason(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
+13
View File
@@ -54,6 +54,13 @@ pub const Task = struct {
// before the reap records it for `process_exit_reason`. Meaningless while
// the task lives.
exit_reason: abi.ExitReason = .exited,
// Endpoint this process's signals arrive on (signal_bind), or null — same
// ownership rules as exit_endpoint (holds a reference; opaque here).
signal_endpoint: ?*anyopaque = null,
// Signals posted but not yet delivered: the coalescing pending mask
// (docs/process-lifecycle.md). Bits are abi.Signal values. Signals pend here
// until an endpoint is bound; two pending terminates are one terminate.
pending_signals: u32 = 0,
// Set by process_kill on a task that is running on another core; the kernel
// finishes the kill at that task's next system call or timer tick.
kill_pending: bool = false,
@@ -647,9 +654,15 @@ fn reapKillPendingLocked() void {
/// other critical section, but releases it *without* touching the interrupt flag
/// — the handler's `iretq` restores the interrupted context's flags, so
/// re-enabling here would open a nested-interrupt window before the return.
/// Called from the tick with the big kernel lock held — process.zig hangs the
/// one-shot timer sweep here (timer_bind), the same call-up pattern as the
/// teardown hooks below.
pub var timer_tick_hook: ?*const fn () void = null;
pub fn tick() void {
_ = sync.enter();
wakeExpired();
if (timer_tick_hook) |hook| hook();
reapKillPendingLocked();
if (preemption_enabled) schedule();
sync.leaveIsr();
+49
View File
@@ -136,6 +136,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
claimReleaseTest(boot_information);
} else if (eql(case, "vfs-client-death")) {
vfsClientDeathTest(boot_information);
} else if (eql(case, "signals")) {
signalsTest(boot_information);
} else if (eql(case, "initial-ramdisk")) {
initialRamdiskTest(boot_information);
} else if (eql(case, "vfs")) {
@@ -1605,6 +1607,53 @@ fn vfsClientDeathTest(boot_information: *const BootInformation) void {
result();
}
/// M17.4 from ring 3: process-test's signal-run role drives the whole lifecycle
/// surface — the zero-length ping (answered by the harness), signals as
/// statements (reload logged, terminate = clean exit), the one-shot timer, and
/// both endings of the stop sequence (polite -> exited, deaf -> killed at the
/// deadline). Its "process-test: signals ok" is the pass marker.
fn signalsTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: signals\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
process.setInitialRamdisk(image); // the parent system_spawns its children by name
process.write_count = 0;
var runner: u32 = 0;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(item.name, "process-test")) continue;
runner = process.spawnProcessSupervised(item.blob, 4, &.{ "process-test", "signal-run" }, scheduler.currentId(), null) catch 0;
break;
}
check("signal-run parent spawned", runner != 0);
const pass_marker = "process-test: signals ok";
const fail_marker = "process-test: FAIL";
scheduler.setPriority(1);
const deadline = architecture.millis() + 15000;
var saw_pass = false;
var saw_fail = false;
while (architecture.millis() < deadline and !saw_pass and !saw_fail) {
if (process.write_len >= pass_marker.len and eql(process.write_buffer[0..pass_marker.len], pass_marker)) saw_pass = true;
if (process.write_len >= fail_marker.len and eql(process.write_buffer[0..fail_marker.len], fail_marker)) saw_fail = true;
scheduler.yield();
}
scheduler.setPriority(4);
check("the signal-run parent reported ok", saw_pass and !saw_fail);
result();
}
/// The whole user-side surface at once: spawn process-test's supervisor role,
/// which — entirely from ring 3 — creates an exit endpoint, spawns its two
/// children supervised, sees them in process_enumerate, kills them (one blocked,