display: resilient scanout — supervised driver, survive loss, re-attach (v2 V6)
The compositor now survives the virtio-gpu driver dying and re-attaches when device-manager restarts it — the last piece of display v2. Three parts: - The driver hellos the device manager (role: bus). It never did, so the manager — which spawns it supervised and expects a hello — was stopping it at the 3s hello deadline every run (the gate markers just printed first). Now it is properly supervised: not stopped for silence, and restarted on death. - A kernel IPC fix so a call to a dead service errors instead of hanging. An endpoint records its owner; when that task dies, its registered endpoints are marked dead (and any parked senders woken with -EPEER), so ipc_call returns -EPEER rather than blocking on a reply that will never come. Without this the compositor's first present after the driver died blocked forever. General robustness — any client of any service benefits. - The compositor re-attaches. Its .scanout calls now fail cleanly (caught), freezing the last frame; when the restarted driver re-announces, attach_scanout detects the backend is already virtio and logs "scanout re-attached", mapping the fresh shared surface and re-looking-up .scanout. (The previous shm mapping leaks — no shm_unmap syscall yet — but its frames are the dead driver's, reclaimed on exit.) - device-manager gains a "test-scanout-restart" mode (like test-usb-restart) that kills the virtio-gpu driver once after it hellos; the displayReattachTest kernel scenario drives it. Gate: python3 test/qemu_test.py display-reattach — "scanout upgraded to virtio-gpu" then "scanout re-attached", no CPU exception, passing 3/3. host tests, ipc/ipc-call/ipc-cap, supervision, shm, display-service, display-demo, virtio-gpu, display-native, and display-modeset all pass; default zig build clean. v2 (V1-V6) complete.
This commit is contained in:
@@ -90,6 +90,10 @@ const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
/// (per process) and/or by a registry slot, counted by `refcount`.
|
||||
pub const Endpoint = struct {
|
||||
refcount: u32 = 1,
|
||||
// The task that created it. When that task dies, the endpoint is marked `dead` so a caller
|
||||
// gets -EPEER instead of blocking forever on a service that will never reply again (V6).
|
||||
owner: u32 = 0,
|
||||
dead: bool = false,
|
||||
// Callers blocked in `call`, awaiting receive, in FIFO order (threaded via
|
||||
// Task.next; each such task is .blocked and in no scheduler queue).
|
||||
sender_head: ?*Task = null,
|
||||
@@ -110,10 +114,30 @@ pub const Endpoint = struct {
|
||||
|
||||
pub fn createIpcEndpoint() ?*Endpoint {
|
||||
const endpoint = heap.allocator().create(Endpoint) catch return null;
|
||||
endpoint.* = .{};
|
||||
endpoint.* = .{ .owner = scheduler.currentId() };
|
||||
return endpoint;
|
||||
}
|
||||
|
||||
/// A task is dying: kill the endpoints it registered as services. Mark each `dead` (so a later
|
||||
/// `call` returns -EPEER rather than blocking on a reply that will never come), wake anyone
|
||||
/// already parked sending to it with that error, and vacate its registry slot. Only *registered*
|
||||
/// endpoints are reachable from here; unregistered ones drop with the task's handle table. The
|
||||
/// caller holds the big kernel lock (this runs on the death path). See docs/display-v2.md (V6).
|
||||
pub fn killOwnedEndpointsLocked(task_id: u32) void {
|
||||
for (®istry) |*slot| {
|
||||
const endpoint = slot.* orelse continue;
|
||||
if (endpoint.owner != task_id) continue;
|
||||
endpoint.dead = true;
|
||||
while (dequeueSender(endpoint)) |sender| {
|
||||
sender.ipc_status = -EPEER;
|
||||
sender.ipc_received_cap = abi.no_cap;
|
||||
scheduler.readyLocked(sender);
|
||||
}
|
||||
slot.* = null;
|
||||
dropRef(endpoint);
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop a reference; free the endpoint when the last one goes. (Frames are leaked
|
||||
/// today like other kernel objects — but the refcount bookkeeping lands now.)
|
||||
pub fn dropRef(endpoint: *Endpoint) void {
|
||||
@@ -293,6 +317,7 @@ pub fn call(endpoint: *Endpoint, message_ptr: u64, message_len: u64, reply_ptr:
|
||||
if (message_len > MESSAGE_MAXIMUM or reply_cap > MESSAGE_MAXIMUM) return -E2BIG;
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (endpoint.dead) return -EPEER; // the service that owned this endpoint is gone — don't block
|
||||
|
||||
const me = scheduler.current();
|
||||
me.ipc_send_ptr = message_ptr;
|
||||
|
||||
@@ -737,6 +737,7 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
scheduler.readyLocked(client); // its blocked `call` now returns the error
|
||||
}
|
||||
ipc.abandonSenderLocked(t);
|
||||
ipc.killOwnedEndpointsLocked(t.id); // its registered services are gone: callers get -EPEER, not a hang
|
||||
scheduler.removeFromWaitQueueLocked(t);
|
||||
scheduler.forgetIpcClientLocked(t);
|
||||
ipc.closeHandles(t);
|
||||
|
||||
@@ -107,6 +107,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
virtioGpuTest(boot_information);
|
||||
} else if (eql(case, "display-native")) {
|
||||
displayNativeTest(boot_information);
|
||||
} else if (eql(case, "display-reattach")) {
|
||||
displayReattachTest(boot_information);
|
||||
} else if (eql(case, "clock")) {
|
||||
clockTest();
|
||||
} else if (eql(case, "smp")) {
|
||||
@@ -2504,6 +2506,50 @@ fn displayNativeTest(boot_information: *const BootInformation) void {
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// V6 — resilience: the compositor survives the virtio-gpu driver dying and re-attaches when
|
||||
/// device-manager restarts it (docs/display-v2.md). Same boot as display-native, but the
|
||||
/// manager runs in "test-scanout-restart" mode: a moment after the driver hellos, it kills it
|
||||
/// once; the normal restart policy respawns it, the restarted driver re-announces, and the
|
||||
/// compositor re-attaches to the fresh scanout — logging `display: scanout re-attached` after
|
||||
/// the initial `display: scanout upgraded to virtio-gpu`. The compositor must not crash.
|
||||
fn displayReattachTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: display-reattach\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{ "device-manager", "test-scanout-restart" }, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
if (manager == 0) {
|
||||
log("display-reattach: could not spawn device-manager\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
if (!spawnNamed(rd, "display")) {
|
||||
log("display-reattach: could not spawn the display service\n", .{});
|
||||
result();
|
||||
return;
|
||||
}
|
||||
_ = spawnNamed(rd, "display-demo");
|
||||
scheduler.setPriority(1); // below the compositor, the demo, and the driver, so they run
|
||||
while (true) scheduler.yield();
|
||||
}
|
||||
|
||||
/// Process arguments, end to end: spawn args-echo bare (its argv[0] is the
|
||||
/// initial-ramdisk name). Instance 1 sees argc == 1 and respawns itself through
|
||||
/// `system_spawn` with the extra arguments "alpha beta-42" — the syscall argument
|
||||
|
||||
Reference in New Issue
Block a user