Publish exit events to subscribers; the VFS releases dead clients' handles (M17.3)
process_subscribe adds an endpoint to a bounded, ref-counted subscriber table; every death posts the same badge encoding a supervisor's exit notification uses, equally late, so subscribers observe a fully-released child. A dying subscriber's own subscriptions are removed first — it never hears about itself. The VFS is the first subscriber: open handles now record their owner and are swept when the owner dies, because a service must never depend on clients cleaning up after themselves (docs/process-lifecycle.md). Proven by the vfs-client-death scenario.
This commit is contained in:
@@ -201,6 +201,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.process_enumerate => systemProcessEnumerate(state),
|
||||
.process_kill => systemProcessKill(state),
|
||||
.process_exit_reason => systemProcessExitReason(state),
|
||||
.process_subscribe => systemProcessSubscribe(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -574,6 +575,16 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
recordExitLocked(t);
|
||||
irq.releaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
// A dead subscriber's own subscriptions go first: it must not hear about
|
||||
// itself, and the slots' endpoint references drop with it.
|
||||
for (&exit_subscribers) |*slot| {
|
||||
if (slot.*) |subscriber| {
|
||||
if (subscriber.owner == t.id) {
|
||||
ipc.dropRef(subscriber.endpoint);
|
||||
slot.* = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (t.ipc_client) |client| {
|
||||
t.ipc_client = null;
|
||||
client.ipc_status = -ipc.EPEER;
|
||||
@@ -583,6 +594,12 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
scheduler.removeFromWaitQueueLocked(t);
|
||||
scheduler.forgetIpcClientLocked(t);
|
||||
ipc.closeHandles(t);
|
||||
// Publish the exit to every subscriber (docs/process-lifecycle.md): the same
|
||||
// badge encoding as the supervisor's notification, and equally late, so a
|
||||
// subscriber also observes a fully-released child.
|
||||
for (&exit_subscribers) |*slot| {
|
||||
if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | t.id);
|
||||
}
|
||||
if (t.exit_endpoint) |raw| {
|
||||
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
|
||||
t.exit_endpoint = null;
|
||||
@@ -690,6 +707,34 @@ pub fn exitReasonOf(caller_id: u32, target_id: u32) i64 {
|
||||
return -ipc.ESRCH;
|
||||
}
|
||||
|
||||
/// The published exit events' subscribers (docs/process-lifecycle.md "Who learns
|
||||
/// of a death"): stateful services — the VFS's file handles, input's
|
||||
/// subscriptions — that must release what a dead client held and cannot learn it
|
||||
/// any other way (a client that simply never calls again looks like silence).
|
||||
/// Bounded like every kernel table; each entry holds its own endpoint reference.
|
||||
const exit_subscriber_capacity = 8;
|
||||
const ExitSubscriber = struct { endpoint: *ipc.Endpoint, owner: u32 };
|
||||
var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exit_subscriber_capacity;
|
||||
|
||||
/// process_subscribe(endpoint): subscribe the caller's endpoint to published exit
|
||||
/// events. Ungated, like process_enumerate — what is running (and dying) is not a
|
||||
/// secret between cooperating processes. -ENOSPC when the table is full.
|
||||
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
for (&exit_subscribers) |*slot| {
|
||||
if (slot.* == null) {
|
||||
endpoint.refcount += 1; // the slot's own reference, dropped on unsubscribe-by-death
|
||||
slot.* = .{ .endpoint = endpoint, .owner = t.id };
|
||||
return architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
}
|
||||
failErr(state, ipc.ENOSPC);
|
||||
}
|
||||
|
||||
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
|
||||
@@ -134,6 +134,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
supervisionTest(boot_information);
|
||||
} else if (eql(case, "claim-release")) {
|
||||
claimReleaseTest(boot_information);
|
||||
} else if (eql(case, "vfs-client-death")) {
|
||||
vfsClientDeathTest(boot_information);
|
||||
} else if (eql(case, "initial-ramdisk")) {
|
||||
initialRamdiskTest(boot_information);
|
||||
} else if (eql(case, "vfs")) {
|
||||
@@ -1535,6 +1537,74 @@ fn claimReleaseTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// M17.3: the published exit events, proven by their first subscriber. The VFS
|
||||
/// subscribes at startup; a client opens a file and parks holding the handle;
|
||||
/// the kill posts the exit event to the VFS's endpoint; the VFS releases the
|
||||
/// dead client's handle and says so — the service-side mirror of iron rule 1
|
||||
/// (a service must never depend on clients cleaning up after themselves).
|
||||
fn vfsClientDeathTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: vfs-client-death\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
check("vfs spawned", spawnNamed(rd, "vfs"));
|
||||
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
var client: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "vfs-test")) continue;
|
||||
client = process.spawnProcessSupervised(item.blob, 4, &.{ "vfs-test", "park" }, me, endpoint) catch 0;
|
||||
break;
|
||||
}
|
||||
check("parked client spawned (supervised)", client != 0);
|
||||
|
||||
// Its heartbeat is the fence: once it beats, the handle is open.
|
||||
const parked = "vfstest: parked";
|
||||
scheduler.setPriority(1);
|
||||
var deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("client parked holding an open handle", process.write_len >= parked.len and eql(process.write_buffer[0..parked.len], parked));
|
||||
|
||||
check("the kill is accepted", process.killProcess(me, client) == 0);
|
||||
var badge: u64 = 0;
|
||||
var received_cap: u64 = 0;
|
||||
_ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||
check("the exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | client);
|
||||
|
||||
// The VFS heard the same published event; its release line is the proof.
|
||||
const released = "vfs: released 1 handle(s) for dead client";
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= released.len and eql(process.write_buffer[0..released.len], released)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("the VFS released the dead client's handle", process.write_len >= released.len and eql(process.write_buffer[0..released.len], released));
|
||||
result();
|
||||
}
|
||||
|
||||
/// The whole user-side surface at once: spawn process-test's supervisor role,
|
||||
/// which — entirely from ring 3 — creates an exit endpoint, spawns its two
|
||||
/// children supervised, sees them in process_enumerate, kills them (one blocked,
|
||||
|
||||
Reference in New Issue
Block a user