display: damage-rect list + tile-grid trackers, vectorizable pixel loops, wide WC stores

Tearing mitigation for the GOP floor, attacking the copy window from three sides:

- Damage is no longer one bounding box. Two trackers, A/B-switchable at
  compile time (display.zig damage_mode): DamageList (free-form dirty rects,
  overlap-merged) and TileGrid (fixed 64-px tiles, exact O(1) marking, runs
  coalesced back into rects). Far-apart changes — the cursor here, an
  animating layer there — no longer unite into one huge repaint.
- fillRect/composite/blitTile now work in row spans (@memset/@memcpy), so
  the compiler vectorizes them and ReleaseSafe bounds checks drop to per-row.
- The back->front present streams 8-byte volatile stores (presentSpan);
  Backend.present takes the rect list, so each present copies only what
  changed, faster.

Host tests cover both trackers; the display QEMU cases all pass.
This commit is contained in:
Daniel Samson
2026-07-21 11:22:00 +01:00
parent 981a4af7e0
commit 23f915c593
3 changed files with 370 additions and 53 deletions
+42 -15
View File
@@ -110,22 +110,48 @@ pub const Gop = struct {
};
}
/// Stream the damaged rectangle from the back buffer to the write-combining LFB, row by
/// row (sequential writes — what WC memory wants; the LFB is never read).
pub fn present(self: *const Gop, damage: Rect) void {
const c = damage.intersect(.{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) });
if (c.isEmpty()) return;
var y: i32 = c.y;
while (y < c.bottom()) : (y += 1) {
const off = @as(usize, @intCast(y)) * self.pitch;
const src: [*]const u32 = @ptrCast(@alignCast(self.back + off));
const dst: [*]volatile u32 = @ptrCast(@alignCast(self.front + off));
var x: i32 = c.x;
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
/// Stream each damaged rectangle from the back buffer to the write-combining LFB, row
/// by row (sequential writes — what WC memory wants; the LFB is never read). The rows
/// are copied by `presentSpan` below, which widens the stores by hand: `volatile`
/// keeps the compiler from eliding or reordering framebuffer writes, but it also
/// forbids it from merging them, so a naive per-pixel loop is stuck at one 4-byte
/// store per iteration. Keeping each copy small (the damage list) and each store wide
/// shrinks the window in which scanout can sample a half-written frame.
pub fn present(self: *const Gop, damage: []const Rect) void {
const bounds = Rect{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) };
for (damage) |rect| {
const c = rect.intersect(bounds);
if (c.isEmpty()) continue;
const span: usize = @intCast(c.w);
var y: i32 = c.y;
while (y < c.bottom()) : (y += 1) {
const offset = @as(usize, @intCast(y)) * self.pitch + @as(usize, @intCast(c.x)) * 4;
const source: [*]const u32 = @ptrCast(@alignCast(self.back + offset));
const front_row: [*]volatile u32 = @ptrCast(@alignCast(self.front + offset));
presentSpan(front_row, source, span);
}
}
}
};
/// Copy `count` pixels into the write-combining front buffer with 8-byte volatile stores
/// (plus a 4-byte head/tail where the span isn't 8-aligned — pixel spans are always
/// 4-aligned). The loads come from the cacheable back buffer and are assembled into a
/// `u64` in registers, so nothing here reads the front buffer.
fn presentSpan(destination: [*]volatile u32, source: [*]const u32, count: usize) void {
var i: usize = 0;
if (i < count and (@intFromPtr(destination) & 7) != 0) {
destination[0] = source[0];
i = 1;
}
while (i + 2 <= count) : (i += 2) {
const pair = @as(u64, source[i]) | (@as(u64, source[i + 1]) << 32);
const wide: *volatile u64 = @ptrCast(@alignCast(destination + i));
wide.* = pair;
}
if (i < count) destination[i] = source[i];
}
/// A display mode the native backend can switch to.
pub const Mode = scanout_protocol.Mode;
@@ -152,8 +178,9 @@ pub const VirtioGpu = struct {
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
}
/// Ask the driver to present. The composited pixels are already in the shared surface, so
/// this is a single request over `.scanout`; the driver transfers + fenced-flushes.
pub fn present(self: *const VirtioGpu, damage: Rect) void {
/// this is a single request over `.scanout` regardless of how many damage rectangles
/// accumulated; the driver transfers + fenced-flushes the whole frame.
pub fn present(self: *const VirtioGpu, damage: []const Rect) void {
_ = damage;
var request = scanout_protocol.Request{
.operation = @intFromEnum(scanout_protocol.Operation.present),
@@ -210,7 +237,7 @@ pub const Backend = union(enum) {
inline else => |*b| b.surface(),
};
}
pub fn present(self: *const Backend, damage: Rect) void {
pub fn present(self: *const Backend, damage: []const Rect) void {
switch (self.*) {
inline else => |*b| b.present(damage),
}