display: hot-attach a virtio-gpu native backend over the GOP floor (v2 V4)

The compositor now boots on the GOP framebuffer and upgrades to the virtio-gpu driver
the moment it announces itself — the pluggable-scanout payoff.

The shared surface. The scanout resource is an shm region the driver creates
(shm_physical, a new syscall, hands it the guest-physical for attach_backing) and passes
to the compositor as a capability. The compositor maps it and composites straight into
it: on x86 DMA is cache-coherent, so the cacheable shared pages the CPU paints are exactly
what the device transfers-and-flushes — no copy, no explicit flush.

The handshake. After bring-up the driver looks up .display and sends attach_scanout with
the geometry + the surface capability. The compositor maps the surface, looks up the
driver's .scanout endpoint itself (the driver registered it — no need to pass it), switches
to backend.VirtioGpu, and re-composites the current frame. present() over the native
backend is a present request on .scanout -> transfer-to-host + resource flush. The first
native present is deferred to a one-shot timer: presenting inline from the announce handler
would deadlock, since the driver is still blocked on our reply and not yet serving .scanout.
After it lands, the compositor reads a pixel back from the shared surface to confirm the
frame reached the device's backing.

- shm_physical (syscall 36) + runtime.shm.physical.
- scanout-protocol (the compositor->driver present channel), separate from the
  client-facing display protocol; the display protocol gains attach_scanout.
- backend.VirtioGpu joins backend.Gop in the tagged union; select() still boots GOP.
- the virtio-gpu driver's scanout backing is now shm (was DMA); it announces + serves
  .scanout present requests (transfer-to-host + flush of the shared surface).

Also fixes a latent framebuffer-geometry corruption the display service hit only when it
enumerated the device tree alongside a busy device-manager: Gop.init now captures the
geometry into a small value the instant device_enumerate returns (rather than re-reading
the 328-byte descriptor across the later claim/mmio_map syscalls) and retries on a zero
geometry. The underlying device-table clobber is a separate kernel bug, tracked apart.

Gate: python3 test/qemu_test.py display-native (QEMU -device virtio-gpu-pci) — "display:
scanout upgraded to virtio-gpu" + "display: native present verified" + "display-demo: ok",
passing 3/3. host tests, display-service, display-demo, shm, and virtio-gpu still pass.
This commit is contained in:
Daniel Samson
2026-07-14 12:21:37 +01:00
parent 6e0e0a62c6
commit 58927ed7e5
13 changed files with 379 additions and 58 deletions
+89 -29
View File
@@ -18,11 +18,18 @@ const runtime = @import("runtime");
const mmio = @import("mmio");
const device = runtime.device;
const dma = runtime.dma;
const shm = runtime.shm;
const system = runtime.system;
const ipc = runtime.ipc;
const dp = runtime.display_protocol;
const sp = runtime.scanout_protocol;
const vp = @import("virtio-pci.zig");
const vg = @import("virtio-gpu-protocol.zig");
/// The DisplayFormat (device-abi) our B8G8R8X8 scanout resource presents: bgrx = 1. Handed to
/// the compositor in the announce so it packs colours in the surface's byte order.
const display_format_bgrx: u32 = 1;
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
const virtio_vendor: u16 = 0x1AF4;
@@ -59,10 +66,15 @@ var notify_addr: usize = 0;
// be asked to map the same resource twice.
var bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 };
// DMA memory: the virtqueue rings, the command scratch, and the scanout backing.
// DMA memory: the virtqueue rings and the command scratch.
var ring: dma.Region = undefined;
var command: dma.Region = undefined;
var backing: dma.Region = undefined;
// The scanout backing is a **shared** (shm) region, not DMA: cacheable so the compositor
// composites into it cheaply (x86 DMA is coherent, so the device still sees the writes), and
// shareable so the same physical pages the device scans out of are the ones the compositor
// paints. The driver keeps the capability to hand to the compositor in the announce.
var surface: shm.Region = undefined;
// Split-virtqueue producer/consumer shadows.
var avail_shadow: u16 = 0;
@@ -335,8 +347,14 @@ fn initialise(endpoint: ipc.Handle) bool {
}
}
backing = dma.alloc(scanout_bytes, dma.coherent) orelse {
log("virtio-gpu: scanout backing allocation failed\n", .{});
// Back the resource with a shared (shm) surface, so the compositor and the device work
// the same physical pages. The device needs the guest-physical base for attach_backing.
surface = shm.create(scanout_bytes) orelse {
log("virtio-gpu: scanout surface allocation failed\n", .{});
return false;
};
const surface_physical = shm.physical(surface.handle) orelse {
log("virtio-gpu: could not resolve the scanout surface physical address\n", .{});
return false;
};
{
@@ -347,7 +365,7 @@ fn initialise(endpoint: ipc.Handle) bool {
.nr_entries = 1,
};
const entry: *vg.MemEntry = @ptrFromInt(command.virtual + request_offset + @sizeOf(vg.ResourceAttachBacking));
entry.* = .{ .addr = backing.physical, .length = @intCast(scanout_bytes) };
entry.* = .{ .addr = surface_physical, .length = @intCast(scanout_bytes) };
if (command_nodata(@sizeOf(vg.ResourceAttachBacking) + @sizeOf(vg.MemEntry)) != ok_nodata) {
log("virtio-gpu: resource_attach_backing failed\n", .{});
return false;
@@ -368,12 +386,35 @@ fn initialise(endpoint: ipc.Handle) bool {
}
log("virtio-gpu: scanout {d}x{d} online\n", .{ scanout_width, scanout_height });
// Paint a known pattern, transfer it to the host resource, and flush it to the display.
const pixels: [*]u32 = @ptrFromInt(backing.virtual);
// Paint a known pattern, present it, and read it back — the V3 self-test that proves the
// whole path (virtqueue, resource, shared backing, transfer, flush) before a client attaches.
const pixels: [*]u32 = @ptrCast(@alignCast(surface.ptr));
const pixel_count: u32 = scanout_width * scanout_height;
for (0..pixel_count) |i| pixels[i] = testPixel(@intCast(i));
mmio.wmb();
if (!presentFull()) {
log("virtio-gpu: initial present failed\n", .{});
return false;
}
// The scanout surface is CPU-visible RAM: read the pattern back to prove the mapping,
// which together with the flush ack above is the automated stand-in for "it's on screen".
mmio.rmb();
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(pixel_count / 2)) {
log("virtio-gpu: pixel read-back mismatch\n", .{});
return false;
}
log("virtio-gpu: flush acked, pixel check ok\n", .{});
// Offer the shared surface to the compositor so it upgrades off the GOP floor (V4).
announce();
return true;
}
/// Present the whole surface: copy the guest backing into the host resource, then flush it to
/// the panel. Reused by the V3 self-test and by every compositor present over `.scanout`. V4
/// presents the full surface; the damage-rect fast path is a later refinement.
fn presentFull() bool {
mmio.wmb(); // the surface writes must be visible before the device transfers them
{
const request = requestAt(vg.TransferToHost2d);
request.* = .{
@@ -382,10 +423,7 @@ fn initialise(endpoint: ipc.Handle) bool {
.offset = 0,
.resource_id = resource_id,
};
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) {
log("virtio-gpu: transfer_to_host_2d failed\n", .{});
return false;
}
if (command_nodata(@sizeOf(vg.TransferToHost2d)) != ok_nodata) return false;
}
{
const request = requestAt(vg.ResourceFlush);
@@ -394,30 +432,52 @@ fn initialise(endpoint: ipc.Handle) bool {
.rect = .{ .x = 0, .y = 0, .width = scanout_width, .height = scanout_height },
.resource_id = resource_id,
};
if (command_nodata(@sizeOf(vg.ResourceFlush)) != ok_nodata) {
log("virtio-gpu: resource_flush was not acked\n", .{});
return false;
}
if (command_nodata(@sizeOf(vg.ResourceFlush)) != ok_nodata) return false;
}
// The scanout backing is CPU-visible RAM: read the pattern back to prove the mapping,
// which together with the flush ack above is the automated stand-in for "it's on screen".
mmio.rmb();
if (pixels[0] != testPixel(0) or pixels[pixel_count / 2] != testPixel(pixel_count / 2)) {
log("virtio-gpu: pixel read-back mismatch\n", .{});
return false;
}
log("virtio-gpu: flush acked, pixel check ok\n", .{});
return true;
}
/// The `scanout` service exists from V3 so the compositor can find it in V4; it carries no
/// requests yet (the ping is answered by the harness), so anything else is a no-op.
/// Announce the scanout to the display service so it upgrades off the GOP framebuffer: hand it
/// the shared surface as a capability plus the geometry. Best-effort and non-fatal — without a
/// display service (the standalone virtio-gpu bring-up test) the driver is still a valid
/// scanout service; it just serves no one. The display replies immediately (it defers its
/// first present to a timer), so this returns before we start serving `.scanout` — no deadlock.
fn announce() void {
var tries: u32 = 0;
const display = while (tries < 50) : (tries += 1) {
if (ipc.lookup(.display)) |h| break h;
system.sleep(20);
} else {
log("virtio-gpu: no display service to announce to (scanout-only)\n", .{});
return;
};
var request = dp.Request{
.operation = @intFromEnum(dp.Operation.attach_scanout),
.width = scanout_width,
.height = scanout_height,
.colour = display_format_bgrx,
};
var reply: [dp.reply_size]u8 = undefined;
_ = ipc.callCap(display, std.mem.asBytes(&request), &reply, surface.handle) catch {
log("virtio-gpu: announce to display failed\n", .{});
return;
};
log("virtio-gpu: announced scanout to display\n", .{});
}
/// The `.scanout` service: the compositor asks us to put a composited frame on the panel. The
/// pixels are already in the shared surface, so a present is a transfer-to-host + flush.
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
_ = message;
_ = reply;
_ = sender;
_ = capability;
if (message.len < sp.request_size) return 0;
const request = std.mem.bytesToValue(sp.Request, message[0..sp.request_size]);
if (request.operation == @intFromEnum(sp.Operation.present)) {
const presented = presentFull();
const response = sp.Reply{ .status = if (presented) 0 else -1 };
@memcpy(reply[0..sp.reply_size], std.mem.asBytes(&response));
return sp.reply_size;
}
return 0;
}