display: damage-rect list + tile-grid trackers, vectorizable pixel loops, wide WC stores
Tearing mitigation for the GOP floor, attacking the copy window from three sides: - Damage is no longer one bounding box. Two trackers, A/B-switchable at compile time (display.zig damage_mode): DamageList (free-form dirty rects, overlap-merged) and TileGrid (fixed 64-px tiles, exact O(1) marking, runs coalesced back into rects). Far-apart changes — the cursor here, an animating layer there — no longer unite into one huge repaint. - fillRect/composite/blitTile now work in row spans (@memset/@memcpy), so the compiler vectorizes them and ReleaseSafe bounds checks drop to per-row. - The back->front present streams 8-byte volatile stores (presentSpan); Backend.present takes the rect list, so each present copies only what changed, faster. Host tests cover both trackers; the display QEMU cases all pass.
This commit is contained in:
@@ -110,22 +110,48 @@ pub const Gop = struct {
|
||||
};
|
||||
}
|
||||
|
||||
/// Stream the damaged rectangle from the back buffer to the write-combining LFB, row by
|
||||
/// row (sequential writes — what WC memory wants; the LFB is never read).
|
||||
pub fn present(self: *const Gop, damage: Rect) void {
|
||||
const c = damage.intersect(.{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) });
|
||||
if (c.isEmpty()) return;
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const off = @as(usize, @intCast(y)) * self.pitch;
|
||||
const src: [*]const u32 = @ptrCast(@alignCast(self.back + off));
|
||||
const dst: [*]volatile u32 = @ptrCast(@alignCast(self.front + off));
|
||||
var x: i32 = c.x;
|
||||
while (x < c.right()) : (x += 1) dst[@intCast(x)] = src[@intCast(x)];
|
||||
/// Stream each damaged rectangle from the back buffer to the write-combining LFB, row
|
||||
/// by row (sequential writes — what WC memory wants; the LFB is never read). The rows
|
||||
/// are copied by `presentSpan` below, which widens the stores by hand: `volatile`
|
||||
/// keeps the compiler from eliding or reordering framebuffer writes, but it also
|
||||
/// forbids it from merging them, so a naive per-pixel loop is stuck at one 4-byte
|
||||
/// store per iteration. Keeping each copy small (the damage list) and each store wide
|
||||
/// shrinks the window in which scanout can sample a half-written frame.
|
||||
pub fn present(self: *const Gop, damage: []const Rect) void {
|
||||
const bounds = Rect{ .x = 0, .y = 0, .w = @intCast(self.width), .h = @intCast(self.height) };
|
||||
for (damage) |rect| {
|
||||
const c = rect.intersect(bounds);
|
||||
if (c.isEmpty()) continue;
|
||||
const span: usize = @intCast(c.w);
|
||||
var y: i32 = c.y;
|
||||
while (y < c.bottom()) : (y += 1) {
|
||||
const offset = @as(usize, @intCast(y)) * self.pitch + @as(usize, @intCast(c.x)) * 4;
|
||||
const source: [*]const u32 = @ptrCast(@alignCast(self.back + offset));
|
||||
const front_row: [*]volatile u32 = @ptrCast(@alignCast(self.front + offset));
|
||||
presentSpan(front_row, source, span);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// Copy `count` pixels into the write-combining front buffer with 8-byte volatile stores
|
||||
/// (plus a 4-byte head/tail where the span isn't 8-aligned — pixel spans are always
|
||||
/// 4-aligned). The loads come from the cacheable back buffer and are assembled into a
|
||||
/// `u64` in registers, so nothing here reads the front buffer.
|
||||
fn presentSpan(destination: [*]volatile u32, source: [*]const u32, count: usize) void {
|
||||
var i: usize = 0;
|
||||
if (i < count and (@intFromPtr(destination) & 7) != 0) {
|
||||
destination[0] = source[0];
|
||||
i = 1;
|
||||
}
|
||||
while (i + 2 <= count) : (i += 2) {
|
||||
const pair = @as(u64, source[i]) | (@as(u64, source[i + 1]) << 32);
|
||||
const wide: *volatile u64 = @ptrCast(@alignCast(destination + i));
|
||||
wide.* = pair;
|
||||
}
|
||||
if (i < count) destination[i] = source[i];
|
||||
}
|
||||
|
||||
/// A display mode the native backend can switch to.
|
||||
pub const Mode = scanout_protocol.Mode;
|
||||
|
||||
@@ -152,8 +178,9 @@ pub const VirtioGpu = struct {
|
||||
return .{ .pixels = self.pixels, .stride = self.stride, .width = self.width, .height = self.height };
|
||||
}
|
||||
/// Ask the driver to present. The composited pixels are already in the shared surface, so
|
||||
/// this is a single request over `.scanout`; the driver transfers + fenced-flushes.
|
||||
pub fn present(self: *const VirtioGpu, damage: Rect) void {
|
||||
/// this is a single request over `.scanout` regardless of how many damage rectangles
|
||||
/// accumulated; the driver transfers + fenced-flushes the whole frame.
|
||||
pub fn present(self: *const VirtioGpu, damage: []const Rect) void {
|
||||
_ = damage;
|
||||
var request = scanout_protocol.Request{
|
||||
.operation = @intFromEnum(scanout_protocol.Operation.present),
|
||||
@@ -210,7 +237,7 @@ pub const Backend = union(enum) {
|
||||
inline else => |*b| b.surface(),
|
||||
};
|
||||
}
|
||||
pub fn present(self: *const Backend, damage: Rect) void {
|
||||
pub fn present(self: *const Backend, damage: []const Rect) void {
|
||||
switch (self.*) {
|
||||
inline else => |*b| b.present(damage),
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user